Server definition
- Hash
- sha256:58b048051a082f520837bd86b691ff45c1e73d64250abeeec4e3a9b0aeb40d3f
- What it is
- What a remote MCP server returned when asked what it offers: 14 tools
The blob, as servednamed by its sha256
{
"instructions": "Brainiall Speech API suite with 4 capabilities:\n1. **Brainiall Pronunciation** — assess_pronunciation scores English pronunciation from audio at overall, sentence, word, and phoneme levels (0-100).\n2. **Brainiall Speech** — transcribe_audio converts audio to text with word-level timestamps and confidence scores.\n3. **Brainiall Voice** — synthesize_speech generates natural speech audio from text with 12 English voices, speed control, and WAV output.\n4. **Brainiall Speech Pro** — transcribe_audio_pro provides 99-language transcription with optional speaker diarization.\n\nAll tools accept base64-encoded audio. Read the resources for guides and requirements.",
"tools": [
{
"description": "Assess English pronunciation quality from audio.\n\nScores pronunciation at four levels: overall, sentence, word, and phoneme.\nEach score is 0-100. Phonemes are returned in both IPA and ARPAbet notation.\nSub-300ms inference latency.\n\nArgs:\n audio_base64: Base64-encoded audio data. Supports WAV, MP3, OGG, and WebM formats.\n text: The reference English text that the speaker was expected to read aloud.\n audio_format: Audio format hint — one of 'wav', 'mp3', 'ogg', 'webm'. Defaults to 'wav'.\n\nReturns:\n dict with keys:\n - overallScore (int 0-100): Overall pronunciation quality\n - sentenceScore (int 0-100): Sentence-level fluency and accuracy\n - words (list): Per-word scores, each containing:\n - word (str): The word\n - score (int 0-100): Word pronunciation score\n - phonemes (list): Per-phoneme scores with IPA/ARPAbet notation\n - decodedTranscript (str): What the model heard (ASR transcript)\n - transcript (str): Reference text\n - confidence (float 0-1): Scoring confidence\n - warnings (list[str]): Quality warnings if any\n - audioQuality (dict): Audio metrics (SNR, peak/RMS dB, etc.)",
"inputSchema": {
"properties": {
"audio_base64": {
"description": "Base64-encoded audio data. Supports WAV, MP3, OGG, and WebM formats.",
"maxLength": 20000000,
"type": "string"
},
"audio_format": {
"default": "wav",
"description": "Audio format hint — one of 'wav', 'mp3', 'ogg', 'webm'.",
"type": "string"
},
"text": {
"description": "The reference English text that the speaker was expected to read aloud.",
"maxLength": 10000,
"type": "string"
}
},
"required": [
"audio_base64",
"text"
],
"type": "object"
},
"name": "assess_pronunciation",
"outputSchema": null
},
{
"description": "Check if the Brainiall Pronunciation service is healthy and ready.\n\nReturns:\n dict with keys:\n - status (str): 'healthy' or error state\n - modelLoaded (bool): Whether the scoring model is loaded\n - version (str): API version",
"inputSchema": {
"properties": {},
"type": "object"
},
"name": "check_pronunciation_service",
"outputSchema": null
},
{
"description": "Check if the Brainiall Speech service is healthy and ready.\n\nReturns:\n dict with keys:\n - status (str): 'healthy' or error state\n - modelLoaded (bool): Whether the speech-recognition model is loaded\n - version (str): API version",
"inputSchema": {
"properties": {},
"type": "object"
},
"name": "check_stt_service",
"outputSchema": null
},
{
"description": "Check if the Brainiall Voice service is healthy and ready.\n\nReturns:\n dict with keys:\n - status (str): 'healthy' or error state\n - modelLoaded (bool): Whether the synthesis model is loaded\n - version (str): API version",
"inputSchema": {
"properties": {},
"type": "object"
},
"name": "check_tts_service",
"outputSchema": null
},
{
"description": "Check if the Brainiall Speech Pro service is healthy and ready.\n\nReturns:\n dict with keys:\n - status (str): 'healthy' or error state\n - modelLoaded (bool): Whether the Brainiall Speech Pro engine is loaded\n - diarizeLoaded (bool): Whether the diarization pipeline is loaded\n - version (str): API version\n - modelName (str): Engine version identifier",
"inputSchema": {
"properties": {},
"type": "object"
},
"name": "check_whisper_service",
"outputSchema": null
},
{
"description": "Get the full phoneme inventory supported by the pronunciation scorer.\n\nReturns a list of all English phonemes the engine can assess, including\nARPAbet symbol, IPA equivalent, example word, and phoneme category\n(vowel, consonant, diphthong).\n\nReturns:\n list of dicts, each with keys:\n - arpabet (str): ARPAbet symbol (e.g. 'AA', 'TH')\n - ipa (str): IPA notation\n - example (str): Example word containing the phoneme\n - category (str): vowel, consonant, or diphthong",
"inputSchema": {
"properties": {},
"type": "object"
},
"name": "get_phoneme_inventory",
"outputSchema": null
},
{
"description": "List all available Brainiall Voice synthesis voices with metadata.\n\nReturns:\n dict with keys:\n - voices (list): Available voices, each with id, name, gender, accent, grade\n - defaultVoice (str): Default voice ID",
"inputSchema": {
"properties": {},
"type": "object"
},
"name": "list_tts_voices",
"outputSchema": null
},
{
"description": "Generate natural speech audio from English text.\n\nProduces high-quality speech with 12 English voices.\nReturns base64-encoded WAV audio (16-bit PCM, 24kHz mono) along with metadata.\n\nAvailable voices:\n- af_heart (default), af_bella, af_nicole, af_sarah, af_sky (American female)\n- am_adam, am_michael (American male)\n- bf_emma, bf_isabella (British female)\n- bm_george, bm_lewis, bm_daniel (British male)\n\nArgs:\n text: English text to synthesize (1-5000 characters).\n voice: Voice ID. See list above. Defaults to 'af_heart'.\n speed: Speed multiplier from 0.5 to 2.0 (default: 1.0).\n\nReturns:\n dict with keys:\n - audio_base64 (str): Base64-encoded WAV audio (16-bit PCM, 24kHz)\n - duration_ms (str): Audio duration in milliseconds\n - voice (str): Voice ID used\n - text_length (str): Input text character count\n - processing_ms (str): Synthesis time in milliseconds",
"inputSchema": {
"properties": {
"speed": {
"default": 1,
"description": "Speech speed multiplier (0.5 = half speed, 2.0 = double).",
"type": "number"
},
"text": {
"description": "English text to convert to speech. Max 5000 characters.",
"maxLength": 5000,
"type": "string"
},
"voice": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"default": null,
"description": "Voice ID (e.g. 'af_heart', 'am_adam'). Uses default if omitted."
}
},
"required": [
"text"
],
"type": "object"
},
"name": "synthesize_speech",
"outputSchema": null
},
{
"description": "Transcribe audio to text with word-level timestamps.\n\nConverts spoken English audio into text with optional word-level timestamps\nand per-word confidence scores.\n\nArgs:\n audio_base64: Base64-encoded audio data (WAV, MP3, OGG, FLAC, WebM).\n audio_format: Audio format hint. Auto-detected from magic bytes if omitted.\n include_timestamps: Whether to include word-level timing (default: true).\n\nReturns:\n dict with keys:\n - text (str): Full decoded transcript\n - words (list): Per-word results with timestamps, each containing:\n - word (str): The transcribed word\n - start (float): Start time in seconds\n - end (float): End time in seconds\n - confidence (float 0-1): Word-level confidence\n - audioDurationMs (int): Audio duration in milliseconds\n - metadata (dict): Processing time, audio length, model version\n - audioQuality (dict): Audio metrics (SNR, peak/RMS dB, etc.)",
"inputSchema": {
"properties": {
"audio_base64": {
"description": "Base64-encoded audio data. Supports WAV, MP3, OGG, FLAC, and WebM formats.",
"maxLength": 20000000,
"type": "string"
},
"audio_format": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"default": null,
"description": "Audio format hint — 'wav', 'mp3', 'ogg', 'flac', 'webm'. Auto-detected if omitted."
},
"include_timestamps": {
"default": true,
"description": "If true, include word-level start/end times and confidence.",
"type": "boolean"
}
},
"required": [
"audio_base64"
],
"type": "object"
},
"name": "transcribe_audio",
"outputSchema": null
},
{
"description": "Transcribe audio with Brainiall Speech Pro — multilingual transcription.\n\nSupports 99 languages with automatic language detection, word-level\ntimestamps, per-word confidence scores, and optional speaker diarization\n(identifies who spoke each word). Best-in-class WER (~2%).\n\nArgs:\n audio_base64: Base64-encoded audio (WAV, MP3, OGG, FLAC, WebM).\n language: Language code. Auto-detected if omitted. Supports 99 languages.\n diarize: Enable speaker diarization (default: false). When true, each word\n includes a speaker label (e.g. SPEAKER_00, SPEAKER_01).\n\nReturns:\n dict with keys:\n - text (str): Full decoded transcript\n - words (list): Per-word results with timestamps, each containing:\n - word (str), start (float), end (float), confidence (float 0-1)\n - speaker (str|null): Speaker label when diarize=true\n - speakers (dict|null): Speaker info with count and labels\n - audioDurationMs (int): Audio duration in milliseconds\n - metadata (dict): Processing time, language, languageProbability\n - audioQuality (dict): Audio metrics (SNR, peak/RMS dB, etc.)",
"inputSchema": {
"properties": {
"audio_base64": {
"description": "Base64-encoded audio data. Supports WAV, MP3, OGG, FLAC, and WebM formats.",
"maxLength": 20000000,
"type": "string"
},
"diarize": {
"default": false,
"description": "Enable speaker diarization to identify who spoke each word.",
"type": "boolean"
},
"language": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"default": null,
"description": "Language code (e.g. 'en', 'es', 'zh'). Auto-detected when omitted."
}
},
"required": [
"audio_base64"
],
"type": "object"
},
"name": "transcribe_audio_pro",
"outputSchema": null
},
{
"description": "Enroll a voiceprint for a speaker from ~2s of clear speech. Repeat with more clips to strengthen it.\n\nOnly an irreversible embedding is stored — never the raw audio.\n\nReturns:\n dict with keys: speaker_id (str), n_samples (int), enrolled (bool).",
"inputSchema": {
"properties": {
"audio": {
"description": "Base64-encoded WAV with >= 2s of clear speech",
"type": "string"
},
"group_id": {
"description": "The group/namespace this speaker belongs to",
"maxLength": 256,
"type": "string"
},
"speaker_id": {
"description": "Your identifier for this speaker",
"maxLength": 256,
"type": "string"
}
},
"required": [
"audio",
"speaker_id",
"group_id"
],
"type": "object"
},
"name": "voice_id_enroll",
"outputSchema": null
},
{
"description": "1:N identification — rank everyone enrolled in the group against this clip.\n\nReturns:\n dict with keys: candidates (list of {speaker_id, similarity}, best first).",
"inputSchema": {
"properties": {
"audio": {
"description": "Base64-encoded WAV of the clip to identify",
"type": "string"
},
"group_id": {
"description": "The group/namespace to search within",
"maxLength": 256,
"type": "string"
},
"top_k": {
"default": 5,
"description": "How many candidate speakers to return",
"maximum": 50,
"minimum": 1,
"type": "integer"
}
},
"required": [
"audio",
"group_id"
],
"type": "object"
},
"name": "voice_id_identify",
"outputSchema": null
},
{
"description": "List the speakers enrolled in a group.\n\nReturns:\n dict with keys: speakers (list of {speaker_id, n_samples, ...}).",
"inputSchema": {
"properties": {
"group_id": {
"description": "The group/namespace to list",
"maxLength": 256,
"type": "string"
}
},
"required": [
"group_id"
],
"type": "object"
},
"name": "voice_id_list_speakers",
"outputSchema": null
},
{
"description": "1:1 verification — is this clip the enrolled speaker?\n\nReturns:\n dict with keys: similarity (float), match (bool), threshold (float).",
"inputSchema": {
"properties": {
"audio": {
"description": "Base64-encoded WAV of the clip to check",
"type": "string"
},
"group_id": {
"description": "The group/namespace",
"maxLength": 256,
"type": "string"
},
"speaker_id": {
"description": "The enrolled speaker to verify against",
"maxLength": 256,
"type": "string"
},
"threshold": {
"anyOf": [
{
"type": "number"
},
{
"type": "null"
}
],
"default": null,
"description": "Optional cosine-similarity threshold (defaults to a tuned value); higher = stricter"
}
},
"required": [
"audio",
"speaker_id",
"group_id"
],
"type": "object"
},
"name": "voice_id_verify",
"outputSchema": null
}
]
}Verify it yourself
curl -s https://api.teppi.xyz/v1/evidence/sha256:58b048051a082f520837bd86b691ff45c1e73d64250abeeec4e3a9b0aeb40d3f | sha256sum