2 Commits
Author SHA1 Message Date
alexpolo1andClaude Sonnet 4.6 f9bfd76e00 Upgrade to LiteRT-LM runtime for Gemma 3n E4B support
Release / Build and Release APK (push) Failing after 9m46s
- Add com.google.ai.edge.litertlm:litertlm-android:0.9.0-alpha05 dependency
- New LiteRTModel backend: loads .litertlm files via Engine API with GPU backend
  (CPU fallback if GPU init fails)
- OnDeviceModel.create() priority: LiteRT-LM → MediaPipe → Gemini Nano
- ModelDownloader: updated to Gemma 3n E4B (4.9 GB) and E4B Web (4.3 GB)
  from google/gemma-3n-E4B-it-litert-lm (HuggingFace, license required)
- Download uses Bearer token auth; token saved to SharedPreferences

Co-Authored-By: Claude Sonnet 4.6 <[email protected]>
2026-02-28 22:49:22 +01:00
alexpolo1andClaude Sonnet 4.6 af84d8ebc4 Fix downloads: HuggingFace auth + Gemma 3 1B IT models
Release / Build and Release APK (push) Failing after 11m18s
Google moved all models to HuggingFace (old CDN returns 404).
All downloads now require a HuggingFace API token.

- Add HF token input field (saved to SharedPreferences, restored on relaunch)
- Download uses Bearer auth header
- Switch to Gemma 3 1B IT (litert-community) — .task format, works with
  current MediaPipe API, 555 MB Q4 or 1 GB Q8
- Clear auth error messages (401/403 shown to user)
- Gemma 3n E4B/E2B (.litertlm format) requires runtime upgrade — planned next

Co-Authored-By: Claude Sonnet 4.6 <[email protected]>
2026-02-28 22:36:05 +01:00
7 changed files with 257 additions and 89 deletions
+4 -1
View File
@@ -51,9 +51,12 @@ dependencies {
// ML Kit GenAI — Gemini Nano via AICore (recommended for Pixel 10)
implementation("com.google.mlkit:genai-prompt:1.0.0-beta1")
// MediaPipe LLM Inference — for custom models (Gemma, etc.)
// MediaPipe LLM Inference — legacy fallback for .task/.bin models
implementation("com.google.mediapipe:tasks-genai:0.10.24")
// LiteRT-LM — primary backend for Gemma 3n .litertlm models
implementation("com.google.ai.edge.litertlm:litertlm-android:0.9.0-alpha05")
// Embedded HTTP server
implementation("org.nanohttpd:nanohttpd:2.3.1")
@@ -0,0 +1,137 @@
package com.pixel10.ai.inference
import android.content.Context
import android.util.Log
import com.google.ai.edge.litertlm.Backend
import com.google.ai.edge.litertlm.Engine
import com.google.ai.edge.litertlm.EngineConfig
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.flow.catch
import kotlinx.coroutines.withContext
import java.io.File
/**
* LiteRT-LM backend for Gemma 3n models (.litertlm format).
*
* This replaces MediaPipe for the newer Gemma 3n E4B/E2B models which use
* the LiteRT-LM runtime. Runs fully on-device using the Tensor G5 GPU.
*
* Model files must be placed in the app's files directory (see [ModelDownloader]).
*/
class LiteRTModel private constructor(
private val engine: Engine,
private val modelName: String
) : OnDeviceModel {
override val backendName = "LiteRT-LM ($modelName)"
@Volatile
override var isReady: Boolean = true
private set
override suspend fun generate(
prompt: String,
maxTokens: Int,
temperature: Float
): String = withContext(Dispatchers.Default) {
val conversation = engine.createConversation()
try {
conversation.sendMessage(prompt).toString()
} catch (e: Exception) {
Log.e(TAG, "LiteRT inference error", e)
throw OnDeviceModel.InferenceException("Generation failed: ${e.message}", e)
} finally {
conversation.close()
}
}
override suspend fun generateStreaming(
prompt: String,
onToken: (String) -> Unit
): String = withContext(Dispatchers.Default) {
val conversation = engine.createConversation()
val sb = StringBuilder()
try {
conversation.sendMessageAsync(prompt)
.catch { e ->
throw OnDeviceModel.InferenceException("Streaming failed: ${e.message}", e)
}
.collect { message ->
val token = message.toString()
sb.append(token)
onToken(token)
}
} finally {
conversation.close()
}
sb.toString()
}
override fun close() {
isReady = false
engine.close()
}
companion object {
private const val TAG = "LiteRTModel"
private val MODEL_EXTENSIONS = listOf("litertlm")
suspend fun create(context: Context): LiteRTModel = withContext(Dispatchers.IO) {
val modelPath = findModelPath(context)
?: throw OnDeviceModel.InferenceException(
"No LiteRT-LM model file found.\n" +
"Download a .litertlm model via the app or place one in:\n" +
" ${context.filesDir.absolutePath}/"
)
val modelName = File(modelPath).name
Log.i(TAG, "Loading LiteRT-LM model: $modelPath")
try {
val config = EngineConfig(
modelPath = modelPath,
backend = Backend.GPU
)
val engine = Engine(config)
withContext(Dispatchers.Default) {
engine.initialize()
}
Log.i(TAG, "LiteRT-LM model loaded: $modelName")
LiteRTModel(engine, modelName)
} catch (gpuError: Exception) {
Log.w(TAG, "GPU backend failed, trying CPU: ${gpuError.message}")
try {
val config = EngineConfig(
modelPath = modelPath,
backend = Backend.CPU
)
val engine = Engine(config)
withContext(Dispatchers.Default) {
engine.initialize()
}
Log.i(TAG, "LiteRT-LM model loaded on CPU: $modelName")
LiteRTModel(engine, modelName)
} catch (e: Exception) {
throw OnDeviceModel.InferenceException(
"Failed to load LiteRT-LM model from $modelPath: ${e.message}", e
)
}
}
}
private fun findModelPath(context: Context): String? {
val searchDirs = listOfNotNull(
context.filesDir,
File(context.filesDir, "models"),
context.getExternalFilesDir(null)
)
for (dir in searchDirs) {
if (!dir.exists()) continue
dir.listFiles()?.firstOrNull { it.extension in MODEL_EXTENSIONS }
?.let { return it.absolutePath }
}
return null
}
}
}
@@ -12,56 +12,46 @@ import java.net.URL
/**
* Downloads a MediaPipe-compatible model for background-safe inference.
*
* Gemini Nano (ML Kit) blocks inference when the app is backgrounded (ErrorCode 30).
* MediaPipe with a local model file has no such restriction — it runs entirely in
* the app process using the Tensor G5 GPU via OpenCL/Vulkan.
* All models are hosted on HuggingFace and require a free API token.
* Get one at: https://huggingface.co/settings/tokens
*
* Three model options (all from Google's MediaPipe CDN):
* - [ModelSpec.GEMMA_3N_E4B_CODING] — best coding/reasoning, ~2.5 GB (recommended)
* - [ModelSpec.GEMMA_3N_E2B_CODING] — good balance, ~1.5 GB
* - [ModelSpec.GEMMA_2B_GENERAL] — lightest, ~1.3 GB
*
* Custom models (DeepSeek Coder, Qwen2.5-Coder, etc.) can be placed manually in
* the app's files directory after converting with ai-edge-torch.
* Models use the MediaPipe `.task` format, compatible with [MediaPipeModel].
* Gemma 3n E4B/E2B (`.litertlm` format) requires a runtime upgrade — coming later.
*/
object ModelDownloader {
private const val TAG = "ModelDownloader"
private const val HF_BASE = "https://huggingface.co"
/** Available model specs that can be downloaded from Google's MediaPipe CDN. */
/** Available model specs downloadable from HuggingFace (requires token + license acceptance). */
enum class ModelSpec(
val displayName: String,
val filename: String,
val url: String,
val repo: String,
val sizeMb: Int,
val description: String
) {
/** Recommended: best coding & reasoning quality via MoE architecture. */
GEMMA_3N_E4B_CODING(
/**
* Gemma 3n E4B INT4 — best quality, Tensor G5 optimised, background-safe.
* Accept license at: https://huggingface.co/google/gemma-3n-E4B-it-litert-lm
*/
GEMMA_3N_E4B(
displayName = "Gemma 3n E4B",
filename = "gemma-3n-E4B-it-int4.task",
url = "https://storage.googleapis.com/mediapipe-models/llm_inference/" +
"gemma-3n-E4B-it-int4/float16/1/gemma-3n-E4B-it-int4.task",
sizeMb = 2500,
description = "Best coding & reasoning (~2.5 GB)"
filename = "gemma-3n-E4B-it-int4.litertlm",
repo = "google/gemma-3n-E4B-it-litert-lm",
sizeMb = 4920,
description = "Best quality — Tensor G5 optimised (~4.9 GB)"
),
/** Good balance between quality and speed. */
GEMMA_3N_E2B_CODING(
displayName = "Gemma 3n E2B",
filename = "gemma-3n-E2B-it-int4.task",
url = "https://storage.googleapis.com/mediapipe-models/llm_inference/" +
"gemma-3n-E2B-it-int4/float16/1/gemma-3n-E2B-it-int4.task",
sizeMb = 1500,
description = "Good balance, faster (~1.5 GB)"
),
/** Lightest option — general-purpose, not optimised for code. */
GEMMA_2B_GENERAL(
displayName = "Gemma 2B",
filename = "gemma-2b-it-gpu-int4.bin",
url = "https://storage.googleapis.com/mediapipe-models/llm_inference/" +
"gemma-2b-it-gpu-int4/float16/1/gemma-2b-it-gpu-int4.bin",
sizeMb = 1300,
description = "Lightest, general-purpose (~1.3 GB)"
/**
* Gemma 3n E4B Web INT4 — smaller variant, slightly lower quality.
* Same license as above.
*/
GEMMA_3N_E4B_WEB(
displayName = "Gemma 3n E4B (Web)",
filename = "gemma-3n-E4B-it-int4-Web.litertlm",
repo = "google/gemma-3n-E4B-it-litert-lm",
sizeMb = 4280,
description = "Slightly smaller variant (~4.3 GB)"
)
}
@@ -82,35 +72,47 @@ object ModelDownloader {
fun modelFile(context: Context, spec: ModelSpec): File =
File(context.filesDir, spec.filename)
/** Legacy compat — returns the file of the installed model, or Gemma 3n E4B path as default. */
/** Returns the file of the installed model, or E4B path as default. */
fun modelFile(context: Context): File =
installedSpec(context)?.let { modelFile(context, it) }
?: modelFile(context, ModelSpec.GEMMA_3N_E4B_CODING)
?: modelFile(context, ModelSpec.GEMMA_3N_E4B)
/**
* Download [spec], reporting progress via [onProgress].
* Download [spec] from HuggingFace, using [hfToken] for authentication.
* Supports resume — if a partial file exists, continues from where it left off.
*
* Get a free token at https://huggingface.co/settings/tokens
*/
suspend fun download(
context: Context,
spec: ModelSpec = ModelSpec.GEMMA_3N_E4B_CODING,
spec: ModelSpec = ModelSpec.GEMMA_3N_E4B,
hfToken: String,
onProgress: (Progress) -> Unit
) = withContext(Dispatchers.IO) {
if (hfToken.isBlank()) throw OnDeviceModel.InferenceException(
"HuggingFace token required.\nGet a free token at huggingface.co/settings/tokens"
)
val dest = modelFile(context, spec)
val alreadyDownloaded = if (dest.exists()) dest.length() else 0L
val downloadUrl = "$HF_BASE/${spec.repo}/resolve/main/${spec.filename}"
Log.i(TAG, "Download starting ${spec.displayName} (already have $alreadyDownloaded bytes)")
Log.i(TAG, "Download starting ${spec.displayName} from $downloadUrl (already have $alreadyDownloaded bytes)")
val conn = URL(spec.url).openConnection() as HttpURLConnection
val conn = URL(downloadUrl).openConnection() as HttpURLConnection
try {
conn.connectTimeout = 30_000
conn.readTimeout = 60_000
conn.setRequestProperty("Authorization", "Bearer $hfToken")
if (alreadyDownloaded > 0) {
conn.setRequestProperty("Range", "bytes=$alreadyDownloaded-")
}
conn.connect()
val code = conn.responseCode
if (code == 401 || code == 403) throw OnDeviceModel.InferenceException(
"Authentication failed (HTTP $code).\nCheck your HuggingFace token."
)
val resuming = code == HttpURLConnection.HTTP_PARTIAL // 206
if (code != HttpURLConnection.HTTP_OK && !resuming) {
throw OnDeviceModel.InferenceException("Download failed: HTTP $code")
@@ -112,9 +112,19 @@ interface OnDeviceModel {
* Tap "Download Model" in the app UI to get the MediaPipe model automatically.
*/
suspend fun create(context: Context): OnDeviceModel = withContext(Dispatchers.IO) {
// MediaPipe first — background-safe, GPU-accelerated via Tensor G5
// LiteRT-LM first — Gemma 3n .litertlm format, GPU-accelerated, background-safe
try {
Log.i(TAG, "Attempting MediaPipe LLM with local model...")
Log.i(TAG, "Attempting LiteRT-LM with local .litertlm model...")
val litert = LiteRTModel.create(context)
Log.i(TAG, "LiteRT-LM model ready: ${litert.backendName}")
return@withContext litert
} catch (e: Exception) {
Log.w(TAG, "LiteRT-LM not available: ${e.message}")
}
// MediaPipe fallback — .task/.bin format, background-safe
try {
Log.i(TAG, "Attempting MediaPipe LLM with local .task model...")
val mediapipe = MediaPipeModel.create(context)
Log.i(TAG, "MediaPipe model ready: ${mediapipe.backendName}")
return@withContext mediapipe
@@ -122,7 +132,7 @@ interface OnDeviceModel {
Log.w(TAG, "MediaPipe not available: ${e.message}")
}
// Gemini Nano fallback — only works when app is in foreground
// Gemini Nano last resort — foreground only
try {
Log.i(TAG, "Attempting Gemini Nano via ML Kit (foreground only)...")
val nano = GeminiNanoModel.create(context)
@@ -134,11 +144,10 @@ interface OnDeviceModel {
throw InferenceException(
"No model loaded yet.\n\n" +
"Tap 'Download Model' in the app to download Gemma 2B (~1.3 GB).\n" +
"Tap 'Download Model' in the app to download Gemma 3n E4B.\n" +
"Once downloaded the server works fully in the background.\n\n" +
"Or place a compatible model file in:\n" +
" ${context.filesDir.absolutePath}/\n" +
" Supported: gemma-2b-it-gpu-int4.bin, gemma-3n-E2B.task, etc."
"Or place a .litertlm file in:\n" +
" ${context.filesDir.absolutePath}/"
)
}
}
@@ -5,6 +5,7 @@ import android.content.ComponentName
import android.content.Context
import android.content.Intent
import android.content.ServiceConnection
import android.content.SharedPreferences
import android.content.pm.PackageManager
import android.graphics.drawable.GradientDrawable
import android.net.wifi.WifiManager
@@ -29,6 +30,7 @@ import java.util.Locale
class MainActivity : AppCompatActivity() {
private lateinit var binding: ActivityMainBinding
private lateinit var prefs: SharedPreferences
private var service: ApiServerService? = null
private var bound = false
private var downloading = false
@@ -68,20 +70,23 @@ class MainActivity : AppCompatActivity() {
binding = ActivityMainBinding.inflate(layoutInflater)
setContentView(binding.root)
prefs = getSharedPreferences("pixel10_prefs", MODE_PRIVATE)
requestNotificationPermission()
// Restore saved HF token
binding.etHfToken.setText(prefs.getString("hf_token", ""))
binding.btnToggle.setOnClickListener {
if (service?.isRunning == true) stopServer() else startServer()
}
binding.btnDownloadGemma3nE4b.setOnClickListener {
startModelDownload(ModelSpec.GEMMA_3N_E4B_CODING)
}
binding.btnDownloadGemma3nE2b.setOnClickListener {
startModelDownload(ModelSpec.GEMMA_3N_E2B_CODING)
}
binding.btnDownloadModel.setOnClickListener {
startModelDownload(ModelSpec.GEMMA_2B_GENERAL)
saveHfToken()
startModelDownload(ModelSpec.GEMMA_3N_E4B)
}
binding.btnDownloadGemma3Q8.setOnClickListener {
saveHfToken()
startModelDownload(ModelSpec.GEMMA_3N_E4B_WEB)
}
updateModelCard()
@@ -128,8 +133,18 @@ class MainActivity : AppCompatActivity() {
}
}
private fun saveHfToken() {
val token = binding.etHfToken.text.toString().trim()
prefs.edit().putString("hf_token", token).apply()
}
private fun startModelDownload(spec: ModelSpec) {
if (downloading) return
val token = binding.etHfToken.text.toString().trim()
if (token.isBlank()) {
binding.tvModelDownloadStatus.text = "Enter your HuggingFace token first"
return
}
downloading = true
setDownloadButtonsEnabled(false)
binding.progressDownload.visibility = View.VISIBLE
@@ -137,7 +152,7 @@ class MainActivity : AppCompatActivity() {
lifecycleScope.launch {
try {
ModelDownloader.download(this@MainActivity, spec) { progress ->
ModelDownloader.download(this@MainActivity, spec, token) { progress ->
runOnUiThread {
binding.progressDownload.progress = progress.percent
val mb = progress.downloadedBytes / 1_048_576
@@ -164,24 +179,23 @@ class MainActivity : AppCompatActivity() {
}
private fun setDownloadButtonsEnabled(enabled: Boolean) {
binding.btnDownloadGemma3nE4b.isEnabled = enabled
binding.btnDownloadGemma3nE2b.isEnabled = enabled
binding.btnDownloadModel.isEnabled = enabled
binding.btnDownloadGemma3Q8.isEnabled = enabled
}
private fun updateModelCard() {
val spec = ModelDownloader.installedSpec(this)
if (spec != null) {
binding.tvModelDownloadStatus.text = getString(R.string.model_downloaded, spec.displayName)
binding.btnDownloadGemma3nE4b.visibility = View.GONE
binding.btnDownloadGemma3nE2b.visibility = View.GONE
binding.etHfToken.visibility = View.GONE
binding.btnDownloadModel.visibility = View.GONE
binding.btnDownloadGemma3Q8.visibility = View.GONE
binding.progressDownload.visibility = View.GONE
} else {
binding.tvModelDownloadStatus.text = getString(R.string.model_not_downloaded)
binding.btnDownloadGemma3nE4b.visibility = View.VISIBLE
binding.btnDownloadGemma3nE2b.visibility = View.VISIBLE
binding.etHfToken.visibility = View.VISIBLE
binding.btnDownloadModel.visibility = View.VISIBLE
binding.btnDownloadGemma3Q8.visibility = View.VISIBLE
setDownloadButtonsEnabled(true)
binding.progressDownload.visibility = View.GONE
}
+24 -21
View File
@@ -132,6 +132,19 @@
android:textColor="@color/log_text"
android:textSize="13sp" />
<com.google.android.material.textfield.TextInputEditText
android:id="@+id/etHfToken"
android:layout_width="match_parent"
android:layout_height="48dp"
android:layout_marginTop="8dp"
android:hint="@string/hf_token_hint"
android:inputType="textPassword"
android:textColor="@color/on_surface"
android:textColorHint="@color/log_text"
android:textSize="13sp"
android:fontFamily="monospace"
android:backgroundTint="@color/primary" />
<ProgressBar
android:id="@+id/progressDownload"
style="@android:style/Widget.ProgressBar.Horizontal"
@@ -141,33 +154,23 @@
android:max="100"
android:visibility="gone" />
<com.google.android.material.button.MaterialButton
android:id="@+id/btnDownloadGemma3nE4b"
style="@style/Widget.MaterialComponents.Button.OutlinedButton"
android:layout_width="wrap_content"
android:layout_height="wrap_content"
android:layout_marginTop="8dp"
android:text="@string/btn_download_gemma3n_e4b"
android:textSize="13sp"
app:cornerRadius="8dp" />
<com.google.android.material.button.MaterialButton
android:id="@+id/btnDownloadGemma3nE2b"
style="@style/Widget.MaterialComponents.Button.OutlinedButton"
android:layout_width="wrap_content"
android:layout_height="wrap_content"
android:layout_marginTop="4dp"
android:text="@string/btn_download_gemma3n_e2b"
android:textSize="13sp"
app:cornerRadius="8dp" />
<com.google.android.material.button.MaterialButton
android:id="@+id/btnDownloadModel"
style="@style/Widget.MaterialComponents.Button.OutlinedButton"
android:layout_width="wrap_content"
android:layout_height="wrap_content"
android:layout_marginTop="8dp"
android:text="@string/btn_download_gemma3_q4"
android:textSize="13sp"
app:cornerRadius="8dp" />
<com.google.android.material.button.MaterialButton
android:id="@+id/btnDownloadGemma3Q8"
style="@style/Widget.MaterialComponents.Button.OutlinedButton"
android:layout_width="wrap_content"
android:layout_height="wrap_content"
android:layout_marginTop="4dp"
android:text="@string/btn_download_model"
android:text="@string/btn_download_gemma3_q8"
android:textSize="13sp"
app:cornerRadius="8dp" />
</LinearLayout>
+4 -4
View File
@@ -18,11 +18,11 @@
<string name="model_not_loaded">Model: not loaded</string>
<!-- Controls -->
<string name="btn_download_gemma3n_e4b">⭐ Gemma 3n E4B — Best coding (~2.5 GB)</string>
<string name="btn_download_gemma3n_e2b">Gemma 3n E2B — Faster (~1.5 GB)</string>
<string name="btn_download_model">Gemma 2B — Lightest (~1.3 GB)</string>
<string name="hf_token_hint">HuggingFace token (huggingface.co/settings/tokens)</string>
<string name="btn_download_gemma3_q4">Gemma 3n E4B — Best quality (~4.9 GB)</string>
<string name="btn_download_gemma3_q8">Gemma 3n E4B Web — Smaller (~4.3 GB)</string>
<string name="model_downloaded">✓ %s ready — background inference enabled</string>
<string name="model_not_downloaded">No local model. Download one to enable background inference.</string>
<string name="model_not_downloaded">No local model. Enter HuggingFace token and download.</string>
<string name="btn_start">Start Server</string>
<string name="btn_stop">Stop Server</string>
<string name="port_label">Port:</string>