Fix downloads: HuggingFace auth + Gemma 3 1B IT models
Google moved all models to HuggingFace (old CDN returns 404). All downloads now require a HuggingFace API token. - Add HF token input field (saved to SharedPreferences, restored on relaunch) - Download uses Bearer auth header - Switch to Gemma 3 1B IT (litert-community) — .task format, works with current MediaPipe API, 555 MB Q4 or 1 GB Q8 - Clear auth error messages (401/403 shown to user) - Gemma 3n E4B/E2B (.litertlm format) requires runtime upgrade — planned next Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -12,56 +12,40 @@ import java.net.URL
|
||||
/**
|
||||
* Downloads a MediaPipe-compatible model for background-safe inference.
|
||||
*
|
||||
* Gemini Nano (ML Kit) blocks inference when the app is backgrounded (ErrorCode 30).
|
||||
* MediaPipe with a local model file has no such restriction — it runs entirely in
|
||||
* the app process using the Tensor G5 GPU via OpenCL/Vulkan.
|
||||
* All models are hosted on HuggingFace and require a free API token.
|
||||
* Get one at: https://huggingface.co/settings/tokens
|
||||
*
|
||||
* Three model options (all from Google's MediaPipe CDN):
|
||||
* - [ModelSpec.GEMMA_3N_E4B_CODING] — best coding/reasoning, ~2.5 GB (recommended)
|
||||
* - [ModelSpec.GEMMA_3N_E2B_CODING] — good balance, ~1.5 GB
|
||||
* - [ModelSpec.GEMMA_2B_GENERAL] — lightest, ~1.3 GB
|
||||
*
|
||||
* Custom models (DeepSeek Coder, Qwen2.5-Coder, etc.) can be placed manually in
|
||||
* the app's files directory after converting with ai-edge-torch.
|
||||
* Models use the MediaPipe `.task` format, compatible with [MediaPipeModel].
|
||||
* Gemma 3n E4B/E2B (`.litertlm` format) requires a runtime upgrade — coming later.
|
||||
*/
|
||||
object ModelDownloader {
|
||||
|
||||
private const val TAG = "ModelDownloader"
|
||||
private const val HF_BASE = "https://huggingface.co"
|
||||
|
||||
/** Available model specs that can be downloaded from Google's MediaPipe CDN. */
|
||||
/** Available model specs downloadable from HuggingFace. */
|
||||
enum class ModelSpec(
|
||||
val displayName: String,
|
||||
val filename: String,
|
||||
val url: String,
|
||||
val repo: String,
|
||||
val sizeMb: Int,
|
||||
val description: String
|
||||
) {
|
||||
/** Recommended: best coding & reasoning quality via MoE architecture. */
|
||||
GEMMA_3N_E4B_CODING(
|
||||
displayName = "Gemma 3n E4B",
|
||||
filename = "gemma-3n-E4B-it-int4.task",
|
||||
url = "https://storage.googleapis.com/mediapipe-models/llm_inference/" +
|
||||
"gemma-3n-E4B-it-int4/float16/1/gemma-3n-E4B-it-int4.task",
|
||||
sizeMb = 2500,
|
||||
description = "Best coding & reasoning (~2.5 GB)"
|
||||
/** Recommended: best size/quality trade-off, runs fast on Tensor G5. */
|
||||
GEMMA_3_1B_Q4(
|
||||
displayName = "Gemma 3 1B IT (Q4)",
|
||||
filename = "gemma3-1b-it-int4.task",
|
||||
repo = "litert-community/Gemma3-1B-IT",
|
||||
sizeMb = 555,
|
||||
description = "Best balance — fast & capable (~555 MB)"
|
||||
),
|
||||
/** Good balance between quality and speed. */
|
||||
GEMMA_3N_E2B_CODING(
|
||||
displayName = "Gemma 3n E2B",
|
||||
filename = "gemma-3n-E2B-it-int4.task",
|
||||
url = "https://storage.googleapis.com/mediapipe-models/llm_inference/" +
|
||||
"gemma-3n-E2B-it-int4/float16/1/gemma-3n-E2B-it-int4.task",
|
||||
sizeMb = 1500,
|
||||
description = "Good balance, faster (~1.5 GB)"
|
||||
),
|
||||
/** Lightest option — general-purpose, not optimised for code. */
|
||||
GEMMA_2B_GENERAL(
|
||||
displayName = "Gemma 2B",
|
||||
filename = "gemma-2b-it-gpu-int4.bin",
|
||||
url = "https://storage.googleapis.com/mediapipe-models/llm_inference/" +
|
||||
"gemma-2b-it-gpu-int4/float16/1/gemma-2b-it-gpu-int4.bin",
|
||||
sizeMb = 1300,
|
||||
description = "Lightest, general-purpose (~1.3 GB)"
|
||||
/** Higher quality, slower. Good for complex reasoning. */
|
||||
GEMMA_3_1B_Q8(
|
||||
displayName = "Gemma 3 1B IT (Q8)",
|
||||
filename = "gemma3-1b-it-int8-web.task",
|
||||
repo = "litert-community/Gemma3-1B-IT",
|
||||
sizeMb = 1010,
|
||||
description = "Higher quality, slower (~1 GB)"
|
||||
)
|
||||
}
|
||||
|
||||
@@ -82,35 +66,47 @@ object ModelDownloader {
|
||||
fun modelFile(context: Context, spec: ModelSpec): File =
|
||||
File(context.filesDir, spec.filename)
|
||||
|
||||
/** Legacy compat — returns the file of the installed model, or Gemma 3n E4B path as default. */
|
||||
/** Legacy compat — returns the file of the installed model, or Q4 path as default. */
|
||||
fun modelFile(context: Context): File =
|
||||
installedSpec(context)?.let { modelFile(context, it) }
|
||||
?: modelFile(context, ModelSpec.GEMMA_3N_E4B_CODING)
|
||||
?: modelFile(context, ModelSpec.GEMMA_3_1B_Q4)
|
||||
|
||||
/**
|
||||
* Download [spec], reporting progress via [onProgress].
|
||||
* Download [spec] from HuggingFace, using [hfToken] for authentication.
|
||||
* Supports resume — if a partial file exists, continues from where it left off.
|
||||
*
|
||||
* Get a free token at https://huggingface.co/settings/tokens
|
||||
*/
|
||||
suspend fun download(
|
||||
context: Context,
|
||||
spec: ModelSpec = ModelSpec.GEMMA_3N_E4B_CODING,
|
||||
spec: ModelSpec = ModelSpec.GEMMA_3_1B_Q4,
|
||||
hfToken: String,
|
||||
onProgress: (Progress) -> Unit
|
||||
) = withContext(Dispatchers.IO) {
|
||||
if (hfToken.isBlank()) throw OnDeviceModel.InferenceException(
|
||||
"HuggingFace token required.\nGet a free token at huggingface.co/settings/tokens"
|
||||
)
|
||||
|
||||
val dest = modelFile(context, spec)
|
||||
val alreadyDownloaded = if (dest.exists()) dest.length() else 0L
|
||||
val downloadUrl = "$HF_BASE/${spec.repo}/resolve/main/${spec.filename}"
|
||||
|
||||
Log.i(TAG, "Download starting ${spec.displayName} (already have $alreadyDownloaded bytes)")
|
||||
Log.i(TAG, "Download starting ${spec.displayName} from $downloadUrl (already have $alreadyDownloaded bytes)")
|
||||
|
||||
val conn = URL(spec.url).openConnection() as HttpURLConnection
|
||||
val conn = URL(downloadUrl).openConnection() as HttpURLConnection
|
||||
try {
|
||||
conn.connectTimeout = 30_000
|
||||
conn.readTimeout = 60_000
|
||||
conn.setRequestProperty("Authorization", "Bearer $hfToken")
|
||||
if (alreadyDownloaded > 0) {
|
||||
conn.setRequestProperty("Range", "bytes=$alreadyDownloaded-")
|
||||
}
|
||||
conn.connect()
|
||||
|
||||
val code = conn.responseCode
|
||||
if (code == 401 || code == 403) throw OnDeviceModel.InferenceException(
|
||||
"Authentication failed (HTTP $code).\nCheck your HuggingFace token."
|
||||
)
|
||||
val resuming = code == HttpURLConnection.HTTP_PARTIAL // 206
|
||||
if (code != HttpURLConnection.HTTP_OK && !resuming) {
|
||||
throw OnDeviceModel.InferenceException("Download failed: HTTP $code")
|
||||
|
||||
@@ -5,6 +5,7 @@ import android.content.ComponentName
|
||||
import android.content.Context
|
||||
import android.content.Intent
|
||||
import android.content.ServiceConnection
|
||||
import android.content.SharedPreferences
|
||||
import android.content.pm.PackageManager
|
||||
import android.graphics.drawable.GradientDrawable
|
||||
import android.net.wifi.WifiManager
|
||||
@@ -29,6 +30,7 @@ import java.util.Locale
|
||||
class MainActivity : AppCompatActivity() {
|
||||
|
||||
private lateinit var binding: ActivityMainBinding
|
||||
private lateinit var prefs: SharedPreferences
|
||||
private var service: ApiServerService? = null
|
||||
private var bound = false
|
||||
private var downloading = false
|
||||
@@ -68,20 +70,23 @@ class MainActivity : AppCompatActivity() {
|
||||
binding = ActivityMainBinding.inflate(layoutInflater)
|
||||
setContentView(binding.root)
|
||||
|
||||
prefs = getSharedPreferences("pixel10_prefs", MODE_PRIVATE)
|
||||
requestNotificationPermission()
|
||||
|
||||
// Restore saved HF token
|
||||
binding.etHfToken.setText(prefs.getString("hf_token", ""))
|
||||
|
||||
binding.btnToggle.setOnClickListener {
|
||||
if (service?.isRunning == true) stopServer() else startServer()
|
||||
}
|
||||
|
||||
binding.btnDownloadGemma3nE4b.setOnClickListener {
|
||||
startModelDownload(ModelSpec.GEMMA_3N_E4B_CODING)
|
||||
}
|
||||
binding.btnDownloadGemma3nE2b.setOnClickListener {
|
||||
startModelDownload(ModelSpec.GEMMA_3N_E2B_CODING)
|
||||
}
|
||||
binding.btnDownloadModel.setOnClickListener {
|
||||
startModelDownload(ModelSpec.GEMMA_2B_GENERAL)
|
||||
saveHfToken()
|
||||
startModelDownload(ModelSpec.GEMMA_3_1B_Q4)
|
||||
}
|
||||
binding.btnDownloadGemma3Q8.setOnClickListener {
|
||||
saveHfToken()
|
||||
startModelDownload(ModelSpec.GEMMA_3_1B_Q8)
|
||||
}
|
||||
|
||||
updateModelCard()
|
||||
@@ -128,8 +133,18 @@ class MainActivity : AppCompatActivity() {
|
||||
}
|
||||
}
|
||||
|
||||
private fun saveHfToken() {
|
||||
val token = binding.etHfToken.text.toString().trim()
|
||||
prefs.edit().putString("hf_token", token).apply()
|
||||
}
|
||||
|
||||
private fun startModelDownload(spec: ModelSpec) {
|
||||
if (downloading) return
|
||||
val token = binding.etHfToken.text.toString().trim()
|
||||
if (token.isBlank()) {
|
||||
binding.tvModelDownloadStatus.text = "Enter your HuggingFace token first"
|
||||
return
|
||||
}
|
||||
downloading = true
|
||||
setDownloadButtonsEnabled(false)
|
||||
binding.progressDownload.visibility = View.VISIBLE
|
||||
@@ -137,7 +152,7 @@ class MainActivity : AppCompatActivity() {
|
||||
|
||||
lifecycleScope.launch {
|
||||
try {
|
||||
ModelDownloader.download(this@MainActivity, spec) { progress ->
|
||||
ModelDownloader.download(this@MainActivity, spec, token) { progress ->
|
||||
runOnUiThread {
|
||||
binding.progressDownload.progress = progress.percent
|
||||
val mb = progress.downloadedBytes / 1_048_576
|
||||
@@ -164,24 +179,23 @@ class MainActivity : AppCompatActivity() {
|
||||
}
|
||||
|
||||
private fun setDownloadButtonsEnabled(enabled: Boolean) {
|
||||
binding.btnDownloadGemma3nE4b.isEnabled = enabled
|
||||
binding.btnDownloadGemma3nE2b.isEnabled = enabled
|
||||
binding.btnDownloadModel.isEnabled = enabled
|
||||
binding.btnDownloadGemma3Q8.isEnabled = enabled
|
||||
}
|
||||
|
||||
private fun updateModelCard() {
|
||||
val spec = ModelDownloader.installedSpec(this)
|
||||
if (spec != null) {
|
||||
binding.tvModelDownloadStatus.text = getString(R.string.model_downloaded, spec.displayName)
|
||||
binding.btnDownloadGemma3nE4b.visibility = View.GONE
|
||||
binding.btnDownloadGemma3nE2b.visibility = View.GONE
|
||||
binding.etHfToken.visibility = View.GONE
|
||||
binding.btnDownloadModel.visibility = View.GONE
|
||||
binding.btnDownloadGemma3Q8.visibility = View.GONE
|
||||
binding.progressDownload.visibility = View.GONE
|
||||
} else {
|
||||
binding.tvModelDownloadStatus.text = getString(R.string.model_not_downloaded)
|
||||
binding.btnDownloadGemma3nE4b.visibility = View.VISIBLE
|
||||
binding.btnDownloadGemma3nE2b.visibility = View.VISIBLE
|
||||
binding.etHfToken.visibility = View.VISIBLE
|
||||
binding.btnDownloadModel.visibility = View.VISIBLE
|
||||
binding.btnDownloadGemma3Q8.visibility = View.VISIBLE
|
||||
setDownloadButtonsEnabled(true)
|
||||
binding.progressDownload.visibility = View.GONE
|
||||
}
|
||||
|
||||
@@ -132,6 +132,19 @@
|
||||
android:textColor="@color/log_text"
|
||||
android:textSize="13sp" />
|
||||
|
||||
<com.google.android.material.textfield.TextInputEditText
|
||||
android:id="@+id/etHfToken"
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="48dp"
|
||||
android:layout_marginTop="8dp"
|
||||
android:hint="@string/hf_token_hint"
|
||||
android:inputType="textPassword"
|
||||
android:textColor="@color/on_surface"
|
||||
android:textColorHint="@color/log_text"
|
||||
android:textSize="13sp"
|
||||
android:fontFamily="monospace"
|
||||
android:backgroundTint="@color/primary" />
|
||||
|
||||
<ProgressBar
|
||||
android:id="@+id/progressDownload"
|
||||
style="@android:style/Widget.ProgressBar.Horizontal"
|
||||
@@ -141,33 +154,23 @@
|
||||
android:max="100"
|
||||
android:visibility="gone" />
|
||||
|
||||
<com.google.android.material.button.MaterialButton
|
||||
android:id="@+id/btnDownloadGemma3nE4b"
|
||||
style="@style/Widget.MaterialComponents.Button.OutlinedButton"
|
||||
android:layout_width="wrap_content"
|
||||
android:layout_height="wrap_content"
|
||||
android:layout_marginTop="8dp"
|
||||
android:text="@string/btn_download_gemma3n_e4b"
|
||||
android:textSize="13sp"
|
||||
app:cornerRadius="8dp" />
|
||||
|
||||
<com.google.android.material.button.MaterialButton
|
||||
android:id="@+id/btnDownloadGemma3nE2b"
|
||||
style="@style/Widget.MaterialComponents.Button.OutlinedButton"
|
||||
android:layout_width="wrap_content"
|
||||
android:layout_height="wrap_content"
|
||||
android:layout_marginTop="4dp"
|
||||
android:text="@string/btn_download_gemma3n_e2b"
|
||||
android:textSize="13sp"
|
||||
app:cornerRadius="8dp" />
|
||||
|
||||
<com.google.android.material.button.MaterialButton
|
||||
android:id="@+id/btnDownloadModel"
|
||||
style="@style/Widget.MaterialComponents.Button.OutlinedButton"
|
||||
android:layout_width="wrap_content"
|
||||
android:layout_height="wrap_content"
|
||||
android:layout_marginTop="8dp"
|
||||
android:text="@string/btn_download_gemma3_q4"
|
||||
android:textSize="13sp"
|
||||
app:cornerRadius="8dp" />
|
||||
|
||||
<com.google.android.material.button.MaterialButton
|
||||
android:id="@+id/btnDownloadGemma3Q8"
|
||||
style="@style/Widget.MaterialComponents.Button.OutlinedButton"
|
||||
android:layout_width="wrap_content"
|
||||
android:layout_height="wrap_content"
|
||||
android:layout_marginTop="4dp"
|
||||
android:text="@string/btn_download_model"
|
||||
android:text="@string/btn_download_gemma3_q8"
|
||||
android:textSize="13sp"
|
||||
app:cornerRadius="8dp" />
|
||||
</LinearLayout>
|
||||
|
||||
@@ -18,11 +18,11 @@
|
||||
<string name="model_not_loaded">Model: not loaded</string>
|
||||
|
||||
<!-- Controls -->
|
||||
<string name="btn_download_gemma3n_e4b">⭐ Gemma 3n E4B — Best coding (~2.5 GB)</string>
|
||||
<string name="btn_download_gemma3n_e2b">Gemma 3n E2B — Faster (~1.5 GB)</string>
|
||||
<string name="btn_download_model">Gemma 2B — Lightest (~1.3 GB)</string>
|
||||
<string name="hf_token_hint">HuggingFace token (huggingface.co/settings/tokens)</string>
|
||||
<string name="btn_download_gemma3_q4">⭐ Gemma 3 1B IT Q4 — Fast (~555 MB)</string>
|
||||
<string name="btn_download_gemma3_q8">Gemma 3 1B IT Q8 — Higher quality (~1 GB)</string>
|
||||
<string name="model_downloaded">✓ %s ready — background inference enabled</string>
|
||||
<string name="model_not_downloaded">No local model. Download one to enable background inference.</string>
|
||||
<string name="model_not_downloaded">No local model. Enter HuggingFace token and download.</string>
|
||||
<string name="btn_start">Start Server</string>
|
||||
<string name="btn_stop">Stop Server</string>
|
||||
<string name="port_label">Port:</string>
|
||||
|
||||
Reference in New Issue
Block a user