1#!/usr/bin/env bash 2# Fetch the offline neural voice Whiskers speaks with when the natural voice (ElevenLabs) is 3# unavailable and the tablet's own voice has no data: sherpa-onnx's native library (arm64) and 4# one Piper voice. Idempotent; every download is checked against a pinned sha256. 5# 6# mise run android:fetch-voices 7# 8# Outputs (all gitignored, none ever committed): 9# android/app/src/main/jniLibs/arm64-v8a/libonnxruntime.so, libsherpa-onnx-jni.so 10# android/app/src/main/assets/offline-voice/{model.onnx,tokens.txt,espeak-ng-data/} 11# 12# The Kotlin half (com.k2fsa.sherpa.onnx.Tts.kt) is committed, because it must be the file of 13# the same release as the library (the JNI layer reads its fields by name); this script checks 14# that it still is. A failure here exits non-zero: Gradle ignores that, and the app is then 15# built without the voice and simply does not offer it. Licences: android/NOTICE.md. 16set -euo pipefail 17cd "$(dirname "$0")/.." 18 19ver=v1.13.8 20lib_sha=2ff63469a71cb6009aa2e3ed5f4a670f8abdcbe4bb9ffd23776afc792a6b4f44 # sherpa-onnx-$ver-android.tar.bz2, 46,093,321 bytes 21# The full-precision model, not int8 or fp16. Measured on a Fire HD 10 Kids tablet (MT8169, 2026-10-04): the int8 model took 11 to 22# 14 s to load and made speech no faster than it plays; the fp32 one loads in about 4 s and runs at a fifth of real 23# time; the fp16 one is half the size but sherpa-onnx 1.13.8's runtime refuses to load it ("Type (tensor(float16)) of 24# output arg (/enc_p/Cast_1_output_0) ... does not match expected type (tensor(float))"). 25voice=vits-piper-en_US-ljspeech-medium 26voice_sha=3dfb4b759d8be032a4903a9538d128b0fda2a06ab1de6cbc2d93a97e2dd83dba # 67,169,893 bytes 27tts_kt_sha=1a2a19e247553f774c5abdfda8319ce9e1a44e21279b16dc2c63e963a82ea9b1 # sherpa-onnx/kotlin-api/Tts.kt at $ver 28 29kt=android/app/src/main/kotlin/com/k2fsa/sherpa/onnx/Tts.kt 30echo "$tts_kt_sha $kt" | sha256sum -c - >&2 \ 31 || { echo "$kt is not the Tts.kt of sherpa-onnx $ver: the library and its Kotlin half must match" >&2; exit 1; } 32 33jni=android/app/src/main/jniLibs/arm64-v8a 34assets=android/app/src/main/assets/offline-voice 35tmp=$(mktemp -d) 36trap 'rm -rf "$tmp"' EXIT 37 38if [ ! -f "$jni/libsherpa-onnx-jni.so" ] || [ ! -f "$jni/libonnxruntime.so" ]; then 39 curl -sSL -o "$tmp/lib.tar.bz2" "https://github.com/k2-fsa/sherpa-onnx/releases/download/$ver/sherpa-onnx-$ver-android.tar.bz2" 40 echo "$lib_sha $tmp/lib.tar.bz2" | sha256sum -c - >&2 41 tar -xjf "$tmp/lib.tar.bz2" -C "$tmp" ./jniLibs/arm64-v8a/libonnxruntime.so ./jniLibs/arm64-v8a/libsherpa-onnx-jni.so 42 mkdir -p "$jni" 43 cp "$tmp"/jniLibs/arm64-v8a/libonnxruntime.so "$tmp"/jniLibs/arm64-v8a/libsherpa-onnx-jni.so "$jni"/ 44fi 45 46if [ ! -f "$assets/model.onnx" ] || [ ! -f "$assets/tokens.txt" ] || [ ! -d "$assets/espeak-ng-data" ]; then 47 curl -sSL -o "$tmp/voice.tar.bz2" "https://github.com/k2-fsa/sherpa-onnx/releases/download/tts-models/$voice.tar.bz2" 48 echo "$voice_sha $tmp/voice.tar.bz2" | sha256sum -c - >&2 49 tar -xjf "$tmp/voice.tar.bz2" -C "$tmp" 50 rm -rf "$assets" 51 mkdir -p "$assets" 52 cp "$tmp/$voice/en_US-ljspeech-medium.onnx" "$assets/model.onnx" 53 cp "$tmp/$voice/tokens.txt" "$assets/tokens.txt" 54 cp -r "$tmp/$voice/espeak-ng-data" "$assets/espeak-ng-data" 55 # The dictionaries of the other 100-odd languages (11 MB, Russian alone 8.5) are never read by an English voice. 56 find "$assets/espeak-ng-data" -maxdepth 1 -name '*_dict' ! -name en_dict -delete 57fi