diff --git a/.github/workflows/kernelrace-ci.yml b/.github/workflows/kernelrace-ci.yml new file mode 100644 index 0000000..47ae20d --- /dev/null +++ b/.github/workflows/kernelrace-ci.yml @@ -0,0 +1,63 @@ +name: KernelRace CI + +on: + pull_request: + paths: + - "KernelRace/**" + - "skainet-ui/**" + - ".github/workflows/kernelrace-ci.yml" + push: + branches: + - develop + paths: + - "KernelRace/**" + - "skainet-ui/**" + - ".github/workflows/kernelrace-ci.yml" + +concurrency: + group: kernelrace-ci-${{ github.ref }} + cancel-in-progress: true + +jobs: + build-and-test: + runs-on: ubuntu-latest + + steps: + - name: Checkout repo + uses: actions/checkout@v4 + + - name: Set up JDK + uses: actions/setup-java@v4 + with: + distribution: temurin + java-version: "21" + + - name: Set up Gradle + uses: gradle/actions/setup-gradle@v4 + + - name: Make gradlew executable + run: chmod +x KernelRace/gradlew + + # composeApp's Kotlin compile tasks depend on the bundled GGUF (fetchModel — see + # composeApp/build.gradle.kts) even for a plain compile check, since Compose resource + # processing needs the file present. Cache it so only the first run per model version + # pays the ~145 MB download. + - name: Cache bundled GGUF model + uses: actions/cache@v4 + with: + path: KernelRace/composeApp/src/commonMain/composeResources/files/SmolLM2-135M-Instruct-Q8_0.gguf + key: kernelrace-smollm2-q8_0-gguf-v1 + + # Unit tests on JVM + Android (no device), the Android debug APK assembly (proves the + # JNI kernel packaging), and a wasm compile check. Deliberately skips instrumented tests + # (needs a real device) to keep this a fast PR check. + - name: Build and test + working-directory: KernelRace + run: | + ./gradlew \ + :shared:jvmTest \ + :shared:testDebugUnitTest \ + :composeApp:assembleDebug \ + :shared:compileKotlinWasmJs \ + :composeApp:compileKotlinWasmJs \ + --console=plain diff --git a/AndroidNeonLlmDemo/.gitignore b/AndroidNeonLlmDemo/.gitignore deleted file mode 100644 index 761be67..0000000 --- a/AndroidNeonLlmDemo/.gitignore +++ /dev/null @@ -1,4 +0,0 @@ -.gradle/ -build/ -local.properties -app/src/main/assets/*.gguf diff --git a/AndroidNeonLlmDemo/README.md b/AndroidNeonLlmDemo/README.md deleted file mode 100644 index 74b1540..0000000 --- a/AndroidNeonLlmDemo/README.md +++ /dev/null @@ -1,50 +0,0 @@ -# AndroidNeonLlmDemo - -On-device LLM chat on Android, accelerated by SKaiNET's hand-written ARM NEON -kernels — with a built-in **NEON vs scalar** comparison so you can measure the -difference on your own phone. - -SmolLM2-135M (Q8_0 GGUF) decodes fully on-device through -`skainet-backend-jni-cpu`: hand-written NEON matmul kernels behind a JNI -bridge, with two `.so` tiers (`armv8-a` and `armv8.2-a+fp16+dotprod`) selected -at runtime per device. - -## What it demonstrates - -- **~5-line integration**: `DecoderGgufWeightLoader` → `OptimizedLLMRuntime` → - `generateUntilStop`, streaming tokens into Compose (see `LlmEngine.kt`). -- **NEON | SCALAR switch**: two chips re-pin the kernel registry (engine - reloads on the next run) — same APK, same model, full-device A/B with a live - tok/s counter. After every generation the kernel time breakdown is logged - under the `SKAINET_DEMO` tag. -- **Split-screen race**: one button launches a second process with the scalar - provider pinned and starts both generations simultaneously. - - ![Split-screen race: NEON at 44.7 tok/s vs scalar at 9.3 tok/s](docs/screenshots/split_race.png) -- **Model delivery**: downloads the GGUF from the Hugging Face Hub on first - run (SKaiNET's Ktor fetcher, streamed to disk with progress). To go fully - offline instead, place `SmolLM2-135M-Instruct-Q8_0.gguf` in - `app/src/main/assets/` before building. -- **SKaiNET design system** from the shared [`../skainet-ui`](../skainet-ui) - module (theme, logo, `FadingRingLoader`). - -## Run - -Real ARM64 hardware shows the point best (an x86 emulator falls back to scalar): - -```sh -./gradlew :app:installDebug -``` - -Tap **Generate on-device** — the model (~145 MB) downloads on first use. Then -flip the **SCALAR** chip and generate again to see the difference, or tap -**Race against scalar** for the side-by-side version. - -Reference numbers (SmolLM2-135M Q8_0, SKaiNET 0.39.1): ~6.4× decode-kernel -throughput NEON vs scalar on a Pixel 8a; generation is matmul-bound since -0.39.1's eager-op fast paths. - -## Requirements - -- ARM64 Android device (minSdk 24, one APK covers armv8.0 through armv9) -- JDK 17+ for the build; no NDK needed (kernels ship inside the AAR) diff --git a/AndroidNeonLlmDemo/app/build.gradle.kts b/AndroidNeonLlmDemo/app/build.gradle.kts deleted file mode 100644 index 1217864..0000000 --- a/AndroidNeonLlmDemo/app/build.gradle.kts +++ /dev/null @@ -1,75 +0,0 @@ -import java.util.Properties - -plugins { - alias(libs.plugins.android.application) - alias(libs.plugins.kotlin.android) - alias(libs.plugins.compose.compiler) -} - -android { - namespace = "sk.ainet.demo" - compileSdk = 36 - - defaultConfig { - applicationId = "sk.ainet.demo" - minSdk = 24 - targetSdk = 36 - versionCode = 1 - versionName = "1.0" - // ARM only — the JNI NEON kernels are the whole point of the demo. - ndk { abiFilters += "arm64-v8a" } - - // Optional Hugging Face token for gated repos. Lives in local.properties - // (gitignored) as HF_TOKEN=hf_xxx — never in source, never committed. - val hfToken = Properties().apply { - rootProject.file("local.properties").takeIf { it.exists() }?.inputStream()?.use(::load) - }.getProperty("HF_TOKEN") ?: "" - buildConfigField("String", "HF_TOKEN", "\"$hfToken\"") - // Shown next to the logo — always matches the dependency in the catalog. - buildConfigField("String", "SKAINET_VERSION", "\"${libs.versions.skainet.asProvider().get()}\"") - } - - buildFeatures { - compose = true - buildConfig = true - } - compileOptions { - sourceCompatibility = JavaVersion.VERSION_11 - targetCompatibility = JavaVersion.VERSION_11 - } -} - -kotlin { - compilerOptions { - jvmTarget.set(org.jetbrains.kotlin.gradle.dsl.JvmTarget.JVM_11) - } -} - -dependencies { - // Shared SKaiNET design system from ../skainet-ui (composite build) - implementation("sk.ainet.ui:skainet-ui") - - implementation(platform(libs.skainet.bom)) - implementation(libs.skainet.lang.core) - implementation(libs.skainet.backend.cpu) - // Hand-written ARM NEON kernels, dispatched at runtime (armv8-a / armv8.2-a+fp16+dotprod) - implementation(libs.skainet.backend.jni.cpu) - // hf:// model download from the Hugging Face Hub - implementation(libs.skainet.data.source) - implementation(libs.skainet.io.core) - implementation(libs.skainet.io.gguf) - - implementation(platform(libs.skainet.transformers.bom)) - implementation(libs.skainet.transformers.core) - implementation(libs.skainet.transformers.runtime.kllama) - implementation(libs.skainet.transformers.inference.llama) - implementation(libs.skainet.transformers.agent) - implementation(libs.kotlinx.io.core) - - implementation(platform(libs.compose.bom)) - implementation(libs.compose.ui) - implementation(libs.compose.material3) - implementation(libs.androidx.activity.compose) - implementation(libs.androidx.lifecycle.viewmodel.compose) - implementation(libs.kotlinx.coroutines.android) -} diff --git a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ChatViewModel.kt b/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ChatViewModel.kt deleted file mode 100644 index a445ebe..0000000 --- a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ChatViewModel.kt +++ /dev/null @@ -1,119 +0,0 @@ -package sk.ainet.demo - -import android.app.Application -import android.os.SystemClock -import androidx.lifecycle.AndroidViewModel -import androidx.lifecycle.viewModelScope -import kotlinx.coroutines.Dispatchers -import kotlinx.coroutines.flow.MutableStateFlow -import kotlinx.coroutines.flow.StateFlow -import kotlinx.coroutines.launch -import kotlinx.coroutines.withContext -import android.util.Log -import sk.ainet.backend.api.kernel.KernelRegistry -import sk.ainet.exec.kernel.jni.JniKernelProvider -import sk.ainet.exec.kernel.jni.JniKernels -import sk.ainet.exec.tensor.ops.KernelProfile - -data class UiState( - val status: String = "Model not loaded", - val kernelTier: String = "", - val output: String = "", - val tokensPerSecond: Double? = null, - val busy: Boolean = false, - val scalarMode: Boolean = false, -) - -/** - * NEON tier label from hardware capability (dispatched .so tier), NOT from - * KernelRegistry — the registry is mutated by mode switches (a scalar-mode - * run leaves it as [scalar] until the next load), which made the NEON chip - * mislabel itself "SCALAR" after switching back. - */ -private fun neonTierLabel(): String = - if (JniKernelProvider.isAvailable()) "ARM NEON" else "NEON (unavailable)" - -/** Cross-process race trigger: main window broadcasts, scalar window listens. */ -const val ACTION_RACE = "sk.ainet.demo.action.RACE" -const val EXTRA_PROMPT = "prompt" - -class ChatViewModel(app: Application) : AndroidViewModel(app) { - - private val _state = MutableStateFlow( - UiState(kernelTier = if ((app as SkainetDemoApp).isScalarProcess) "SCALAR" else neonTierLabel()) - ) - val state: StateFlow = _state - - /** - * Fullscreen A/B switch for clean screen recordings: the whole phone runs - * either the NEON or the scalar kernels — no split-screen CPU sharing. - * Reloads the model on the next generate (kernel lookups are cached per - * execution context). - */ - fun setKernelMode(scalar: Boolean) { - if (_state.value.busy || _state.value.scalarMode == scalar) return - viewModelScope.launch { - getApplication().setKernelMode(if (scalar) KernelMode.SCALAR else KernelMode.NEON) - _state.value = _state.value.copy( - scalarMode = scalar, - kernelTier = if (scalar) "SCALAR" else neonTierLabel(), - output = "", - tokensPerSecond = null, - status = "Kernel mode: ${if (scalar) "scalar" else "NEON"} — model reloads on next run", - ) - } - } - - /** Load the engine (download + weights) without generating — arms a fair race. */ - fun preload() { - if (_state.value.busy) return - _state.value = _state.value.copy(busy = true) - viewModelScope.launch { - try { - getApplication().engine { progress -> - _state.value = _state.value.copy(status = progress) - } - _state.value = _state.value.copy(busy = false, status = "Model loaded — ready to race") - } catch (e: Exception) { - _state.value = _state.value.copy(busy = false, status = "Error: ${e.message ?: e.javaClass.simpleName}") - } - } - } - - fun generate(prompt: String) { - if (_state.value.busy || prompt.isBlank()) return - _state.value = _state.value.copy(busy = true, output = "", tokensPerSecond = null, status = "Loading model…") - - viewModelScope.launch { - try { - val engine = getApplication().engine { progress -> - _state.value = _state.value.copy(status = progress) - } - _state.value = _state.value.copy(status = "Generating") - KernelProfile.reset() - - var tokenCount = 0 - var firstTokenAt = 0L - withContext(Dispatchers.Default) { - engine.generate(prompt) { piece -> - val now = SystemClock.elapsedRealtime() - if (tokenCount == 0) firstTokenAt = now - tokenCount++ - val elapsedS = (now - firstTokenAt) / 1000.0 - _state.value = _state.value.copy( - output = _state.value.output + piece, - // decode tok/s, measured from the first emitted token (excludes prefill) - tokensPerSecond = if (elapsedS > 0.5) (tokenCount - 1) / elapsedS else null, - ) - } - } - _state.value = _state.value.copy(busy = false, status = "Done — $tokenCount tokens, fully on-device") - // Hard evidence of which kernels actually ran this generation. - Log.i("SKAINET_DEMO_${ProcessTag.suffix}", "providers=${KernelRegistry.availableNames()}") - Log.i("SKAINET_DEMO_${ProcessTag.suffix}", KernelProfile.report()) - } catch (e: Exception) { - _state.value = _state.value.copy(busy = false, status = "Error: ${e.message ?: e.javaClass.simpleName}") - } - } - } -} diff --git a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/LlmEngine.kt b/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/LlmEngine.kt deleted file mode 100644 index 9c2ab28..0000000 --- a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/LlmEngine.kt +++ /dev/null @@ -1,103 +0,0 @@ -package sk.ainet.demo - -import android.os.SystemClock -import android.util.Log -import java.io.File -import sk.ainet.apps.kllama.agent.generateUntilStop -import sk.ainet.apps.llm.InferenceRuntime -import sk.ainet.apps.llm.OptimizedLLMMode -import sk.ainet.apps.llm.OptimizedLLMRuntime -import sk.ainet.apps.llm.Tokenizer -import sk.ainet.apps.llm.tokenizer.TokenizerFactory -import sk.ainet.context.DirectCpuExecutionContext -import sk.ainet.io.AndroidRandomAccessSource -import sk.ainet.io.model.QuantPolicy -import sk.ainet.lang.types.FP32 -import sk.ainet.models.llama.DecoderGgufWeightLoader -import sk.ainet.models.llama.LlamaNetworkLoader - -private val TAG = "SKAINET_DEMO_${ProcessTag.suffix}" - -/** - * Loads a GGUF LLM and streams generated tokens — the whole integration. - * - * The five lines inside [load] + [generate] are the Scene 4 shot of the - * demo video: GGUF in, tokens out, no Python, no C++ in the app build. - * The hand-written ARM NEON kernels (skainet-backend-jni-cpu) register - * themselves through the kernel SPI just by being on the classpath. - */ -class LlmEngine private constructor( - private val runtime: InferenceRuntime, - private val tokenizer: Tokenizer, -) { - - fun generate(prompt: String, maxTokens: Int = 200, onToken: (String) -> Unit): String { - runtime.reset() - // SmolLM2-Instruct is a ChatML model — a raw prompt makes it emit - // <|im_end|> immediately. Wrap in the ChatML envelope it was trained on. - val templated = "<|im_start|>user\n$prompt<|im_end|>\n<|im_start|>assistant\n" - val promptTokens = tokenizer.encode(templated) - PerfLog.event( - "generate_start", - "promptTokens" to promptTokens.size, - "maxTokens" to maxTokens, - "eos" to tokenizer.eosTokenId, - ) - val genStart = SystemClock.elapsedRealtime() - var firstTokenAt = 0L - var tokenCount = 0 - val result = runtime.generateUntilStop( - prompt = promptTokens, - maxTokens = maxTokens, - eosTokenId = tokenizer.eosTokenId, - temperature = 0.7f, - onToken = { tokenId -> - val now = SystemClock.elapsedRealtime() - if (tokenCount == 0) firstTokenAt = now - tokenCount++ - Log.i(TAG, "token #$tokenCount id=$tokenId '${tokenizer.decode(tokenId)}' +${now - genStart}ms") - onToken(tokenizer.decode(tokenId)) - }, - decode = { tokenizer.decode(it) }, - ) - val genEnd = SystemClock.elapsedRealtime() - // Decode speed excludes prefill: measured from the first emitted token. - val decodeS = (genEnd - firstTokenAt) / 1000.0 - PerfLog.event( - "generate_done", - "tokens" to tokenCount, - "ttftMs" to if (tokenCount > 0) firstTokenAt - genStart else null, - "totalMs" to genEnd - genStart, - "decodeTokPerSec" to if (tokenCount > 1 && decodeS > 0) "%.2f".format((tokenCount - 1) / decodeS) else null, - "textLen" to result.text.length, - ) - Log.i(TAG, "result text len=${result.text.length}: '${result.text.take(200)}'") - return result.text - } - - companion object { - - /** Call from Dispatchers.IO — streams weights, never materializes the file on the ART heap. */ - suspend fun load(gguf: File): LlmEngine { - // --- the five lines ------------------------------------------------- - val ctx = DirectCpuExecutionContext.create() - val weights = DecoderGgufWeightLoader( - randomAccessProvider = { AndroidRandomAccessSource.open(gguf.path) }, - quantPolicy = QuantPolicy.NATIVE_OPTIMIZED, // keep Q8_0 packed for the NEON kernels - acceptedArchitectures = setOf("llama", "mistral"), // SmolLM2 is llama-family - ).loadToMapStreaming(ctx) - val runtime = OptimizedLLMRuntime( - model = LlamaNetworkLoader.fromWeights(weights), - ctx = ctx, - mode = OptimizedLLMMode.DIRECT, - dtype = FP32::class, - bos = weights.metadata.bosTokenId, - ) - val tokenizer = AndroidRandomAccessSource.open(gguf.path) - .use { TokenizerFactory.fromGgufSource(it) } - // -------------------------------------------------------------------- - - return LlmEngine(runtime, tokenizer) - } - } -} diff --git a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/MainActivity.kt b/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/MainActivity.kt deleted file mode 100644 index 2b779bc..0000000 --- a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/MainActivity.kt +++ /dev/null @@ -1,180 +0,0 @@ -package sk.ainet.demo - -import android.content.Intent -import android.os.Bundle -import androidx.activity.ComponentActivity -import androidx.activity.compose.setContent -import androidx.compose.foundation.Image -import androidx.compose.foundation.layout.* -import androidx.compose.foundation.rememberScrollState -import androidx.compose.foundation.verticalScroll -import androidx.compose.material3.* -import androidx.compose.runtime.* -import androidx.compose.runtime.saveable.rememberSaveable -import androidx.compose.ui.Alignment -import androidx.compose.ui.Modifier -import androidx.compose.ui.res.painterResource -import androidx.compose.ui.unit.dp -import androidx.lifecycle.viewmodel.compose.viewModel -import sk.ainet.ui.components.indeterminateOrbitingFadingRingLoader -import sk.ainet.ui.theme.SKaiNETTheme - -class MainActivity : ComponentActivity() { - override fun onCreate(savedInstanceState: Bundle?) { - super.onCreate(savedInstanceState) - setContent { SKaiNETTheme { ChatScreen() } } - } -} - -@OptIn(ExperimentalMaterial3Api::class) -@Composable -fun ChatScreen(viewModel: ChatViewModel = viewModel()) { - val state by viewModel.state.collectAsState() - var prompt by remember { mutableStateOf("Explain what a NEON (ARM) instruction is, in two sentences.") } - - Scaffold( - topBar = { - TopAppBar( - title = { - Row(verticalAlignment = Alignment.CenterVertically) { - Image( - painter = painterResource(R.drawable.skainet_logo), - contentDescription = "SKaiNET logo", - modifier = Modifier.size(36.dp), - ) - Spacer(Modifier.width(10.dp)) - Text("SKaiNET", style = MaterialTheme.typography.titleLarge) - Spacer(Modifier.width(6.dp)) - Text( - BuildConfig.SKAINET_VERSION, - style = MaterialTheme.typography.labelSmall, - color = MaterialTheme.colorScheme.onSurfaceVariant, - modifier = Modifier.align(Alignment.Bottom).padding(bottom = 4.dp), - ) - } - }, - actions = { - state.tokensPerSecond?.let { - Text( - "%.1f tok/s".format(it), - style = MaterialTheme.typography.titleLarge, - color = MaterialTheme.colorScheme.primary, - modifier = Modifier.padding(end = 16.dp), - ) - } - }, - ) - }, - ) { padding -> - Column( - Modifier.fillMaxSize().padding(padding) - .padding(horizontal = 16.dp, vertical = 4.dp) - ) { - val context0 = androidx.compose.ui.platform.LocalContext.current - val inScalarProcess = context0.applicationContext - .let { it as? SkainetDemoApp }?.isScalarProcess == true - if (inScalarProcess) { - AssistChip(onClick = {}, label = { Text(state.kernelTier) }) - } else { - // Fullscreen A/B switch: whole phone on one kernel path — - // record one run per mode and compare. - Row(verticalAlignment = Alignment.CenterVertically) { - FilterChip( - selected = !state.scalarMode, - onClick = { viewModel.setKernelMode(false) }, - enabled = !state.busy, - label = { Text(if (state.scalarMode) "NEON" else state.kernelTier) }, - ) - Spacer(Modifier.width(8.dp)) - FilterChip( - selected = state.scalarMode, - onClick = { viewModel.setKernelMode(true) }, - enabled = !state.busy, - label = { Text("SCALAR") }, - ) - } - } - // Status on its own line — with large font scales it doesn't fit - // next to the chips and used to blow up the row height. - Text( - state.status, - style = MaterialTheme.typography.bodySmall, - maxLines = 2, - modifier = Modifier.fillMaxWidth().padding(top = 2.dp), - ) - Spacer(Modifier.height(6.dp)) - - val loading = state.busy && state.output.isEmpty() - if (loading) { - // SKaiNET fading-ring loader while the model downloads/loads - Column( - Modifier.weight(1f).fillMaxWidth(), - horizontalAlignment = Alignment.CenterHorizontally, - verticalArrangement = Arrangement.Center, - ) { - indeterminateOrbitingFadingRingLoader(size = 120.dp) - Spacer(Modifier.height(16.dp)) - Text(state.status, style = MaterialTheme.typography.bodyMedium) - } - } else { - Text( - text = state.output.ifEmpty { "…" }, - modifier = Modifier.weight(1f).fillMaxWidth().verticalScroll(rememberScrollState()), - style = MaterialTheme.typography.bodyLarge, - ) - } - - Spacer(Modifier.height(8.dp)) - OutlinedTextField( - value = prompt, - onValueChange = { prompt = it }, - modifier = Modifier.fillMaxWidth(), - label = { Text("Prompt") }, - ) - Spacer(Modifier.height(8.dp)) - Button( - onClick = { viewModel.generate(prompt) }, - enabled = !state.busy, - modifier = Modifier.fillMaxWidth(), - ) { - Text(if (state.busy) "Generating…" else "Generate on-device") - } - - val context = androidx.compose.ui.platform.LocalContext.current - val isMainProcess = context.applicationContext - .let { it as? SkainetDemoApp }?.isScalarProcess == false - if (isMainProcess) { - Spacer(Modifier.height(4.dp)) - // Saveable: entering split-screen recreates the Activity, and plain - // remember{} would silently disarm the race button. - var raceArmed by rememberSaveable { mutableStateOf(false) } - // Tap 1: open the scalar twin adjacent and preload BOTH engines. - // Tap 2: broadcast the prompt so both windows start in the same instant. - TextButton( - onClick = { - if (!raceArmed) { - context.startActivity( - Intent(context, ScalarActivity::class.java).addFlags( - Intent.FLAG_ACTIVITY_LAUNCH_ADJACENT or Intent.FLAG_ACTIVITY_NEW_TASK - ) - ) - viewModel.preload() - raceArmed = true - } else { - context.sendBroadcast( - Intent(ACTION_RACE) - .setPackage(context.packageName) - .putExtra(EXTRA_PROMPT, prompt) - ) - viewModel.generate(prompt) - } - }, - enabled = !state.busy, - modifier = Modifier.fillMaxWidth(), - ) { - Text(if (raceArmed) "Start race 🏁" else "Race against scalar (split screen)") - } - } - } - } -} diff --git a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ModelSource.kt b/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ModelSource.kt deleted file mode 100644 index e4c961e..0000000 --- a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ModelSource.kt +++ /dev/null @@ -1,110 +0,0 @@ -package sk.ainet.demo - -import android.content.Context -import android.os.SystemClock -import java.io.File -import java.io.FileNotFoundException -import kotlinx.io.Buffer -import kotlinx.io.buffered -import kotlinx.io.files.SystemFileSystem -import sk.ainet.data.source.KtorRemoteDataSourceFetcher -import kotlinx.io.files.Path as KotlinxPath - -const val HF_REPO = "unsloth/SmolLM2-135M-Instruct-GGUF" -const val HF_FILE = "SmolLM2-135M-Instruct-Q8_0.gguf" - -/** - * Resolves the GGUF file: bundled asset if present (offline demo builds), - * otherwise streamed from the Hugging Face Hub via SKaiNET's Ktor fetcher - * into filesDir. Chunked copy straight to disk — the model never sits on - * the ART heap, and progress is reported per real received bytes. - * (Not the resolver's CachePolicy.Bypass path: that buffers the whole - * artifact in memory before returning — exactly what a 145 MB file on a - * 256 MB heap must avoid.) - * - * Both app processes (NEON and :scalar) share filesDir, so the model is - * fetched once per device. - */ -object ModelSource { - - suspend fun resolve(context: Context, onProgress: (String) -> Unit): File { - val dir = File(context.filesDir, "models").apply { mkdirs() } - val target = File(dir, HF_FILE) - if (target.exists()) { - PerfLog.event("model_cached", "file" to target.name, "fileMB" to target.length() / 1_000_000) - return target - } - - copyFromAssetsOrNull(context, target)?.let { - PerfLog.event("model_from_assets", "file" to it.name, "fileMB" to it.length() / 1_000_000) - return it - } - - return download(target, onProgress) - } - - private fun copyFromAssetsOrNull(context: Context, target: File): File? = try { - context.assets.open(HF_FILE).use { input -> - target.outputStream().use { output -> input.copyTo(output) } - } - target - } catch (_: FileNotFoundException) { - null - } - - private suspend fun download(target: File, onProgress: (String) -> Unit): File { - val url = "https://huggingface.co/$HF_REPO/resolve/main/$HF_FILE" - // HF_TOKEN comes from local.properties via BuildConfig (gated repos only). - val headers = BuildConfig.HF_TOKEN.takeIf { it.isNotBlank() } - ?.let { mapOf("Authorization" to "Bearer $it") } ?: emptyMap() - - // Stream to a .part sibling and rename so an interrupted download - // retries instead of leaving a truncated GGUF behind. - val tmp = File(target.parentFile, target.name + ".part") - val fetcher = KtorRemoteDataSourceFetcher() - PerfLog.event("download_start", "url" to url) - val startedAt = SystemClock.elapsedRealtime() - var received = 0L - try { - val content = fetcher.fetch(url, headers) - val totalMb = content.sizeBytes?.let { "%.0f".format(it / 1e6) } ?: "?" - content.source.use { source -> - SystemFileSystem.sink(KotlinxPath(tmp.path)).buffered().use { sink -> - val chunk = Buffer() - var lastReported = -1L - while (true) { - val n = source.readAtMostTo(chunk, 1024 * 1024) - if (n == -1L) break - sink.write(chunk, n) - received += n - val mb = received / 1_000_000 - if (mb != lastReported) { - lastReported = mb - onProgress("Downloading model… $mb / $totalMb MB") - } - } - } - } - check(tmp.renameTo(target)) { "rename failed: $tmp -> $target" } - val durS = (SystemClock.elapsedRealtime() - startedAt) / 1000.0 - PerfLog.event( - "download_done", - "fileMB" to received / 1_000_000, - "durMs" to (durS * 1000).toLong(), - "MBps" to if (durS > 0) "%.1f".format(received / 1e6 / durS) else null, - ) - return target - } catch (e: Exception) { - tmp.delete() - PerfLog.event( - "download_failed", - "receivedMB" to received / 1_000_000, - "durMs" to SystemClock.elapsedRealtime() - startedAt, - "error" to (e.message ?: e.javaClass.simpleName), - ) - throw e - } finally { - fetcher.close() - } - } -} diff --git a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/PerfLog.kt b/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/PerfLog.kt deleted file mode 100644 index 9e5ddc8..0000000 --- a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/PerfLog.kt +++ /dev/null @@ -1,54 +0,0 @@ -package sk.ainet.demo - -import android.os.Debug -import android.os.SystemClock -import android.util.Log - -/** - * One-line structured perf events for the demo, greppable with: - * - * adb logcat -s SKAINET_PERF_NEON:I SKAINET_PERF_SCALAR:I - * - * The tag carries the process suffix (NEON vs SCALAR — see [ProcessTag]), so - * the one-phone race can be watched as two side-by-side terminal windows: - * - * adb logcat -s SKAINET_PERF_NEON:I # window 1 - * adb logcat -s SKAINET_PERF_SCALAR:I # window 2 - * - * Every line carries the offset since process start plus a memory snapshot - * (ART heap used/max, native heap, total PSS), e.g.: - * - * event=model_load_done | +8123ms | heap=41/256MB | native=163MB | pss=310MB | durMs=6210 | kernels=neon - * - * Logcat itself prefixes the wall-clock timestamp. - */ -object PerfLog { - - val TAG = "SKAINET_PERF_${ProcessTag.suffix}" - - private const val MB = 1024 * 1024L - - private val processStart = SystemClock.elapsedRealtime() - - fun event(name: String, vararg details: Pair) { - val line = buildString { - append("event=").append(name) - append(" | +").append(SystemClock.elapsedRealtime() - processStart).append("ms") - append(" | ").append(memorySnapshot()) - for ((key, value) in details) { - if (value != null) append(" | ").append(key).append('=').append(value) - } - } - Log.i(TAG, line) - } - - /** Reading PSS costs a few ms — fine per event, never call per token. */ - private fun memorySnapshot(): String { - val rt = Runtime.getRuntime() - val heapUsed = (rt.totalMemory() - rt.freeMemory()) / MB - val heapMax = rt.maxMemory() / MB - val native = Debug.getNativeHeapAllocatedSize() / MB - val pssMb = Debug.MemoryInfo().also { Debug.getMemoryInfo(it) }.totalPss / 1024 - return "heap=$heapUsed/${heapMax}MB | native=${native}MB | pss=${pssMb}MB" - } -} diff --git a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ProcessTag.kt b/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ProcessTag.kt deleted file mode 100644 index f9dfd0b..0000000 --- a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ProcessTag.kt +++ /dev/null @@ -1,18 +0,0 @@ -package sk.ainet.demo - -import java.io.File - -/** - * ":scalar" vs main process, read once from /proc/self/cmdline (works pre- - * and post-API 28, no Context needed — usable from plain objects like - * [PerfLog]). Splits log tags so the one-phone NEON-vs-scalar race can be - * watched as two separately filterable `adb logcat` streams. - */ -internal object ProcessTag { - private val NUL_BYTE = 0.toChar() - - val isScalar: Boolean by lazy { - File("/proc/self/cmdline").readText().substringBefore(NUL_BYTE).endsWith(":scalar") - } - val suffix: String by lazy { if (isScalar) "SCALAR" else "NEON" } -} diff --git a/AndroidNeonLlmDemo/build.gradle.kts b/AndroidNeonLlmDemo/build.gradle.kts deleted file mode 100644 index b93419d..0000000 --- a/AndroidNeonLlmDemo/build.gradle.kts +++ /dev/null @@ -1,5 +0,0 @@ -plugins { - alias(libs.plugins.android.application) apply false - alias(libs.plugins.kotlin.android) apply false - alias(libs.plugins.compose.compiler) apply false -} diff --git a/AndroidNeonLlmDemo/gradle.properties b/AndroidNeonLlmDemo/gradle.properties deleted file mode 100644 index 7a9e1c2..0000000 --- a/AndroidNeonLlmDemo/gradle.properties +++ /dev/null @@ -1,2 +0,0 @@ -org.gradle.jvmargs=-Xmx4g -Dfile.encoding=UTF-8 -android.useAndroidX=true diff --git a/AndroidNeonLlmDemo/gradle/libs.versions.toml b/AndroidNeonLlmDemo/gradle/libs.versions.toml deleted file mode 100644 index 9ae011e..0000000 --- a/AndroidNeonLlmDemo/gradle/libs.versions.toml +++ /dev/null @@ -1,36 +0,0 @@ -# Toolchain versions mirror the SKaiNET repos' own catalogs (release/0.39.0) -[versions] -agp = "8.12.3" -kotlin = "2.3.21" -compose-bom = "2026.04.01" -activity-compose = "1.12.4" -lifecycle = "2.9.4" -coroutines = "1.11.0" -skainet = "0.39.1" # official release from Maven Central -skainet-transformers = "0.39.1" # lock-step release against engine 0.39.1 - -[libraries] -skainet-bom = { module = "sk.ainet:skainet-bom", version.ref = "skainet" } -skainet-lang-core = { module = "sk.ainet.core:skainet-lang-core" } -skainet-backend-cpu = { module = "sk.ainet.core:skainet-backend-cpu" } -skainet-backend-jni-cpu = { module = "sk.ainet.core:skainet-backend-jni-cpu" } -skainet-data-source = { module = "sk.ainet.core:skainet-data-source" } -skainet-io-core = { module = "sk.ainet.core:skainet-io-core" } -skainet-io-gguf = { module = "sk.ainet.core:skainet-io-gguf" } -skainet-transformers-bom = { module = "sk.ainet.transformers:skainet-transformers-bom", version.ref = "skainet-transformers" } -skainet-transformers-core = { module = "sk.ainet.transformers:skainet-transformers-core" } -skainet-transformers-runtime-kllama = { module = "sk.ainet.transformers:skainet-transformers-runtime-kllama" } -skainet-transformers-inference-llama = { module = "sk.ainet.transformers:skainet-transformers-inference-llama" } -skainet-transformers-agent = { module = "sk.ainet.transformers:skainet-transformers-agent" } -kotlinx-io-core = { module = "org.jetbrains.kotlinx:kotlinx-io-core", version = "0.9.1" } -androidx-activity-compose = { module = "androidx.activity:activity-compose", version.ref = "activity-compose" } -androidx-lifecycle-viewmodel-compose = { module = "androidx.lifecycle:lifecycle-viewmodel-compose", version.ref = "lifecycle" } -compose-bom = { module = "androidx.compose:compose-bom", version.ref = "compose-bom" } -compose-material3 = { module = "androidx.compose.material3:material3" } -compose-ui = { module = "androidx.compose.ui:ui" } -kotlinx-coroutines-android = { module = "org.jetbrains.kotlinx:kotlinx-coroutines-android", version.ref = "coroutines" } - -[plugins] -kotlin-android = { id = "org.jetbrains.kotlin.android", version.ref = "kotlin" } -android-application = { id = "com.android.application", version.ref = "agp" } -compose-compiler = { id = "org.jetbrains.kotlin.plugin.compose", version.ref = "kotlin" } diff --git a/GloVeEmbeddings/gradle/libs.versions.toml b/GloVeEmbeddings/gradle/libs.versions.toml index fdad168..9ff62d1 100644 --- a/GloVeEmbeddings/gradle/libs.versions.toml +++ b/GloVeEmbeddings/gradle/libs.versions.toml @@ -1,11 +1,11 @@ [versions] -# Must match the Kotlin version SKaiNET was built with: SKaiNET 0.34.0 ships -# klibs compiled with Kotlin 2.3.21, and an older compiler cannot read them +# Must match the Kotlin version SKaiNET was built with: SKaiNET 0.40.1 ships +# klibs compiled with Kotlin 2.4.10, and an older compiler cannot read them # (JS/wasm fail with "Symbol for Any not found"). -kotlin = "2.3.21" +kotlin = "2.4.10" # SKaiNET BOM version — published to Maven Central. The BOM pins every # sk.ainet.core:* artifact so individual libraries are declared without versions. -skainet = "0.34.0" +skainet = "0.40.1" # Compose Multiplatform + Android. Pinned to 1.10.1 to match the shared # :skainet-ui design system (included build): a mismatched Compose pulls two diff --git a/GloVeEmbeddings/gradle/wrapper/gradle-wrapper.properties b/GloVeEmbeddings/gradle/wrapper/gradle-wrapper.properties index 2dcec85..5dd3c01 100644 --- a/GloVeEmbeddings/gradle/wrapper/gradle-wrapper.properties +++ b/GloVeEmbeddings/gradle/wrapper/gradle-wrapper.properties @@ -1,6 +1,6 @@ distributionBase=GRADLE_USER_HOME distributionPath=wrapper/dists -distributionUrl=https\://services.gradle.org/distributions/gradle-9.0-bin.zip +distributionUrl=https\://services.gradle.org/distributions/gradle-9.5.1-bin.zip networkTimeout=10000 validateDistributionUrl=true zipStoreBase=GRADLE_USER_HOME diff --git a/GloVeEmbeddings/settings.gradle.kts b/GloVeEmbeddings/settings.gradle.kts index f17d67d..c2a850b 100644 --- a/GloVeEmbeddings/settings.gradle.kts +++ b/GloVeEmbeddings/settings.gradle.kts @@ -17,7 +17,7 @@ pluginManagement { dependencyResolutionManagement { repositories { - // Resolve everything from Maven Central (SKaiNET 0.34.0 is published there). + // Resolve everything from Maven Central (SKaiNET 0.40.1 is published there). // No mavenLocal: an unrestricted mavenLocal is consulted first and can contain // a partial kotlin-stdlib (JVM jar + POM, no klib variants) that shadows // Central's variant-aware metadata, breaking JS/wasm with "Missing stdlib class". diff --git a/GloVeEmbeddings/webapp.json b/GloVeEmbeddings/webapp.json index ad41dee..4037477 100644 --- a/GloVeEmbeddings/webapp.json +++ b/GloVeEmbeddings/webapp.json @@ -2,6 +2,7 @@ "id": "gloveembeddings", "name": "GloVe Embeddings Explorer", "description": "Offline word-embedding explorer: nearest neighbours and vector analogies (king − man + woman ≈ queen), powered by SKaiNET", + "platforms": ["android", "ios", "desktop", "wasm"], "distDirs": [ "app/build/kotlin-webpack/wasmJs/productionExecutable", "app/build/processedResources/wasmJs/main" diff --git a/KernelRace/.gitignore b/KernelRace/.gitignore new file mode 100644 index 0000000..66c2275 --- /dev/null +++ b/KernelRace/.gitignore @@ -0,0 +1,5 @@ +.gradle/ +build/ +local.properties +composeApp/src/commonMain/composeResources/files/*.gguf +composeApp/src/androidMain/assets/*.gguf diff --git a/KernelRace/.java-version b/KernelRace/.java-version new file mode 100644 index 0000000..5f39e91 --- /dev/null +++ b/KernelRace/.java-version @@ -0,0 +1 @@ +21.0 diff --git a/KernelRace/README.md b/KernelRace/README.md new file mode 100644 index 0000000..932d853 --- /dev/null +++ b/KernelRace/README.md @@ -0,0 +1,107 @@ +# Kernel Race + +On-device LLM chat, accelerated by SKaiNET's hand-written ARM NEON kernels on Android — with a +built-in **NEON vs scalar** comparison so you can measure the difference on your own phone. The +same Kotlin codebase also runs on Desktop and in the browser (WebAssembly), where it shows the +*other* kernel tiers SKaiNET dispatches to when there's no NEON hardware to target. + +SmolLM2-135M (Q8_0 GGUF) decodes fully on-device through `skainet-backend-jni-cpu` on Android: +hand-written NEON matmul kernels behind a JNI bridge, with two `.so` tiers (`armv8-a` and +`armv8.2-a+fp16+dotprod`) selected at runtime per device. + +This example began as the Android-only **AndroidNeonLlmDemo** (still available as a frozen +snapshot at the git tag [`2026_08_arm_android`](../../tree/2026_08_arm_android/AndroidNeonLlmDemo)) +and was converted into a proper Kotlin Multiplatform sample: shared engine/view-model logic, a +Compose Multiplatform UI, and platform-specific runtime construction for Android, Desktop and +Wasm. iOS is deferred until the multiplatform structure is proven out. + +## What it demonstrates + +- **~5-line integration**: `DecoderGgufWeightLoader` → `OptimizedLLMRuntime` → + `generateUntilStop`, streaming tokens into Compose (see `shared/.../engine/LlmEngine.kt` and + the per-platform `LlamaRuntimeBuilder.*.kt` actuals). +- **NEON | SCALAR switch** (Android only): two chips re-pin the kernel registry (engine reloads + on the next run) — same APK, same model, full-device A/B with a live tok/s counter. +- **Split-screen race** (Android only): one button launches a second process with the scalar + provider pinned and starts both generations simultaneously. + + ![Split-screen race: NEON at 44.7 tok/s vs scalar at 9.3 tok/s](docs/screenshots/split_race.png) +- **Cross-platform kernel tiers**: the same Kotlin `LlmEngine` runs on three different kernel + paths — Android's ARM NEON JNI kernels, Desktop's native-optimized file-based load, and Wasm's + in-memory FP32 fallback (browsers have no filesystem, so the model is bundled at build time + instead of downloaded). +- **Model delivery**: Android downloads the GGUF from the Hugging Face Hub on first run + (SKaiNET's Ktor fetcher, streamed to disk with progress) or uses a bundled asset if present; + Desktop downloads to a local cache dir; Wasm bundles the model into the production build via + `scripts/fetch-model.sh` (see the Web section below for the size trade-off this implies). +- **SKaiNET design system** from the shared [`../skainet-ui`](../skainet-ui) module (theme, + logo, `FadingRingLoader`). + +## Run + +### Android (the full experience) + +Real ARM64 hardware shows the point best (an x86 emulator falls back to scalar): + +```sh +./gradlew :composeApp:installDebug +``` + +Tap **Generate on-device** — the model (~145 MB) downloads on first use. Then flip the +**SCALAR** chip and generate again to see the difference, or tap **Race against scalar** for the +side-by-side version. To go fully offline, place `SmolLM2-135M-Instruct-Q8_0.gguf` in +`composeApp/src/androidMain/assets/` before building. + +Reference numbers (SmolLM2-135M Q8_0, SKaiNET 0.39.1): ~6.4× decode-kernel throughput NEON vs +scalar on a Pixel 8a. + +### Desktop + +```sh +./gradlew :composeApp:run +``` + +Downloads the model to `~/.skainet-examples/kernelrace/models/` on first run. No NEON kernels +here — this is SKaiNET's file-based native-optimized load path on plain JVM. + +### Web (Wasm) + +```sh +./gradlew :composeApp:wasmJsBrowserDevelopmentRun +``` + +The first build runs `scripts/fetch-model.sh`, which bundles the ~145 MB GGUF straight into the +production JS/Wasm bundle (browsers have no persistent filesystem to download into at runtime). +That's a deliberately large asset for a web page — acceptable for a local dev run or a +maintainer-approved deploy, but worth knowing about before wiring this into CI or a public +samples page. + +## Requirements + +- Android: ARM64 device (minSdk 24, one APK covers armv8.0 through armv9), JDK 21+ +- Desktop: JDK 21+ with the JDK Vector API (incubator) enabled — wired automatically by the + Gradle build +- Web: a Chromium-based browser for the wasm GC runtime + +## Architecture + +``` +KernelRace/ +├── shared/ # Engine, view-model, model resolution — platform-neutral +│ └── src/ +│ ├── commonMain/ # LlmEngine, ChatViewModel, ModelResolver (pure, unit-tested) +│ ├── androidMain/ # AndroidModelProvider, NEON-aware LlamaRuntimeBuilder actual +│ ├── jvmMain/ # DesktopModelProvider, file-based LlamaRuntimeBuilder actual +│ └── wasmJsMain/ # Bytes-only LlamaRuntimeBuilder actual (no filesystem) +└── composeApp/ # Compose Multiplatform UI + platform entry points + └── src/ + ├── commonMain/ # App/ChatScreen (skainet-ui themed), kernelControls slot + ├── androidMain/ # KernelRaceApp (kernel pinning), race UI, manifest + ├── jvmMain/ # Desktop window entry point + └── wasmJsMain/ # Browser entry point + bundled model resource +``` + +The race mechanics (multi-process kernel pinning, `KernelRegistry`, the split-screen button) are +entirely Android-specific and live in `composeApp/androidMain`. Common code only sees a +`kernelControls` composable slot and a `GenerativeEngine` interface, which is what makes +`shared`'s `ChatViewModel` unit-testable with a fake on every target — see `shared/src/*Test/`. diff --git a/KernelRace/THIRD_PARTY_LICENSES/Apache-2.0.txt b/KernelRace/THIRD_PARTY_LICENSES/Apache-2.0.txt new file mode 100644 index 0000000..d645695 --- /dev/null +++ b/KernelRace/THIRD_PARTY_LICENSES/Apache-2.0.txt @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/KernelRace/THIRD_PARTY_LICENSES/NOTICE b/KernelRace/THIRD_PARTY_LICENSES/NOTICE new file mode 100644 index 0000000..19aba81 --- /dev/null +++ b/KernelRace/THIRD_PARTY_LICENSES/NOTICE @@ -0,0 +1,9 @@ +This product bundles/downloads the SmolLM2-135M-Instruct language model (Q8_0 +quantization) released by Hugging Face (HuggingFaceTB), distributed under the +Apache License 2.0 (see Apache-2.0.txt in this directory). + +Model source: https://huggingface.co/HuggingFaceTB/SmolLM2-135M-Instruct +GGUF build: https://huggingface.co/unsloth/SmolLM2-135M-Instruct-GGUF +GGUF file: SmolLM2-135M-Instruct-Q8_0.gguf (~145 MB) + +Attribution: HuggingFaceTB. See https://github.com/huggingface/smollm. diff --git a/KernelRace/build.gradle.kts b/KernelRace/build.gradle.kts new file mode 100644 index 0000000..58a2e4a --- /dev/null +++ b/KernelRace/build.gradle.kts @@ -0,0 +1,7 @@ +plugins { + alias(libs.plugins.androidApplication) apply false + alias(libs.plugins.androidLibrary) apply false + alias(libs.plugins.kotlinMultiplatform) apply false + alias(libs.plugins.composeMultiplatform) apply false + alias(libs.plugins.composeCompiler) apply false +} diff --git a/KernelRace/composeApp/build.gradle.kts b/KernelRace/composeApp/build.gradle.kts new file mode 100644 index 0000000..10b1033 --- /dev/null +++ b/KernelRace/composeApp/build.gradle.kts @@ -0,0 +1,160 @@ +import java.util.Properties +import org.jetbrains.compose.desktop.application.dsl.TargetFormat +import org.jetbrains.kotlin.gradle.ExperimentalWasmDsl +import org.jetbrains.kotlin.gradle.dsl.JvmTarget + +plugins { + alias(libs.plugins.kotlinMultiplatform) + alias(libs.plugins.androidApplication) + alias(libs.plugins.composeMultiplatform) + alias(libs.plugins.composeCompiler) +} + +kotlin { + jvmToolchain(21) + + androidTarget { + compilerOptions { + jvmTarget.set(JvmTarget.JVM_11) + } + } + + jvm() + + @OptIn(ExperimentalWasmDsl::class) + wasmJs { + browser() + binaries.executable() + } + + sourceSets { + androidMain.dependencies { + implementation(libs.compose.uiTooling) + implementation(libs.androidx.activity.compose) + // Hand-written ARM NEON kernels — Android JNI only. + implementation(libs.skainet.backend.jni.cpu) + } + commonMain.dependencies { + implementation(projects.shared) + implementation("sk.ainet.ui:skainet-ui") + implementation(compose.runtime) + implementation(compose.foundation) + implementation(compose.material3) + implementation(compose.ui) + implementation(compose.components.resources) + implementation(compose.components.uiToolingPreview) + implementation(libs.androidx.lifecycle.viewmodelCompose) + implementation(libs.androidx.lifecycle.runtimeCompose) + implementation(libs.kotlinx.coroutines) + } + commonTest.dependencies { + implementation(libs.kotlin.test) + } + jvmMain.dependencies { + implementation(compose.desktop.currentOs) + implementation(libs.kotlinx.coroutinesSwing) + } + wasmJsMain.dependencies { + implementation(libs.kotlinx.coroutines) + } + } +} + +android { + namespace = "sk.ainet.samples.kernelrace" + compileSdk = libs.versions.android.compileSdk.get().toInt() + + defaultConfig { + applicationId = "sk.ainet.samples.kernelrace" + minSdk = libs.versions.android.minSdk.get().toInt() + targetSdk = libs.versions.android.targetSdk.get().toInt() + versionCode = 1 + versionName = "1.0" + // ARM only — the JNI NEON kernels are the whole point of the race. + ndk { abiFilters += "arm64-v8a" } + + // Optional Hugging Face token for gated repos. Lives in local.properties + // (gitignored) as HF_TOKEN=hf_xxx — never in source, never committed. + // SmolLM2 itself is a public repo, so this is normally left blank. + val hfToken = Properties().apply { + rootProject.file("local.properties").takeIf { it.exists() }?.inputStream()?.use(::load) + }.getProperty("HF_TOKEN") ?: "" + buildConfigField("String", "HF_TOKEN", "\"$hfToken\"") + // Shown next to the logo — always matches the dependency in the catalog. + buildConfigField("String", "SKAINET_VERSION", "\"${libs.versions.skainet.asProvider().get()}\"") + } + + buildFeatures { + compose = true + buildConfig = true + } + packaging { + resources { + excludes += "/META-INF/{AL2.0,LGPL2.1}" + } + } + buildTypes { + getByName("release") { + isMinifyEnabled = false + } + } + compileOptions { + sourceCompatibility = JavaVersion.VERSION_11 + targetCompatibility = JavaVersion.VERSION_11 + } +} + +// Single source-of-truth for the JVM flags SKaiNET needs: the JDK Vector API (incubator) for +// SIMD-accelerated CPU ops on desktop. +val skainetSimdJvmArgs = listOf( + "--add-modules", "jdk.incubator.vector", + "--enable-preview", + "-Dskainet.cpu.vector.enabled=true", +) + +tasks.withType().configureEach { + jvmArgs(skainetSimdJvmArgs) +} + +tasks.withType().configureEach { + jvmArgs(skainetSimdJvmArgs) +} + +val fetchModel by tasks.registering(Exec::class) { + description = "Downloads the SmolLM2-135M-Instruct Q8_0 GGUF into composeResources/files. Skips if present." + group = "build setup" + val scriptPath = rootProject.layout.projectDirectory.file("scripts/fetch-model.sh") + commandLine("bash", scriptPath.asFile.absolutePath) + inputs.file(scriptPath) + val modelFile = rootProject.layout.projectDirectory.file( + "composeApp/src/commonMain/composeResources/files/SmolLM2-135M-Instruct-Q8_0.gguf" + ) + outputs.file(modelFile) + onlyIf { !modelFile.asFile.exists() || modelFile.asFile.length() < 100L * 1024 * 1024 } +} + +// Every Kotlin compile task AND every Compose-resources copy task depends on the model being on +// disk — Gradle 9's strict implicit-dependency validator fails otherwise, since the +// resource-copy task reads from a directory fetchModel writes into. +tasks.matching { + it.name.startsWith("compileKotlin") || + it.name.startsWith("compileJava") || + it.name.startsWith("convertXmlValueResourcesFor") || + it.name.startsWith("copyNonXmlValueResourcesFor") || + it.name.startsWith("prepareComposeResourcesTaskFor") || + it.name.startsWith("generateResourceAccessorsFor") +}.configureEach { dependsOn(fetchModel) } + +compose.desktop { + application { + mainClass = "sk.ainet.samples.kernelrace.MainKt" + + jvmArgs += skainetSimdJvmArgs + + nativeDistributions { + targetFormats(TargetFormat.Dmg, TargetFormat.Msi, TargetFormat.Deb) + packageName = "sk.ainet.samples.kernelrace" + packageVersion = "1.0.0" + } + } +} diff --git a/AndroidNeonLlmDemo/app/src/main/AndroidManifest.xml b/KernelRace/composeApp/src/androidMain/AndroidManifest.xml similarity index 77% rename from AndroidNeonLlmDemo/app/src/main/AndroidManifest.xml rename to KernelRace/composeApp/src/androidMain/AndroidManifest.xml index f395c92..bfe8695 100644 --- a/AndroidNeonLlmDemo/app/src/main/AndroidManifest.xml +++ b/KernelRace/composeApp/src/androidMain/AndroidManifest.xml @@ -1,13 +1,13 @@ - + @@ -23,13 +23,13 @@ + android:taskAffinity="sk.ainet.samples.kernelrace.scalar" /> diff --git a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/SkainetDemoApp.kt b/KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/KernelRaceApp.kt similarity index 55% rename from AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/SkainetDemoApp.kt rename to KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/KernelRaceApp.kt index 76fbc2f..3ebd2b7 100644 --- a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/SkainetDemoApp.kt +++ b/KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/KernelRaceApp.kt @@ -1,39 +1,45 @@ -package sk.ainet.demo +package sk.ainet.samples.kernelrace import android.app.Application import android.content.ComponentCallbacks2 import android.os.Build -import android.os.SystemClock +import java.io.File +import java.util.ServiceLoader import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.sync.Mutex import kotlinx.coroutines.sync.withLock import kotlinx.coroutines.withContext -import java.util.ServiceLoader import sk.ainet.backend.api.kernel.KernelProvider import sk.ainet.backend.api.kernel.KernelRegistry +import sk.ainet.context.DirectCpuExecutionContext import sk.ainet.exec.kernel.ScalarKernelProvider +import sk.ainet.samples.kernelrace.engine.GenerativeEngine +import sk.ainet.samples.kernelrace.engine.LlmEngine +import sk.ainet.samples.kernelrace.model.AndroidModelProvider +import sk.ainet.samples.kernelrace.platform.logEvent + +/** Which kernels the next engine (re)build pins — full-power A/B comparison. */ +enum class KernelMode { NEON, SCALAR } /** - * Application-scoped model holder: the engine is heavyweight, so it lives - * for the app lifetime and is dropped under memory pressure. + * Application-scoped model holder: the engine is heavyweight, so it lives for the app lifetime + * and is dropped under memory pressure. * * Kernel wiring happens here, once per process, before any matmul: - * - main process → ServiceLoader discovers the JNI NEON provider from - * skainet-backend-jni-cpu (highest priority wins). Done directly because - * backend-api's KernelServiceLoader is jvmMain-only — absent from the - * library's Android variant — while java.util.ServiceLoader works on ART. - * - ":scalar" process → [ScalarKernelProvider] is pinned first; every - * dispatch site only auto-installs when the registry is EMPTY, so the - * NEON provider never registers and this process stays scalar. - * This is what powers the one-phone split-screen race in the demo video. + * - main process → ServiceLoader discovers the JNI NEON provider from skainet-backend-jni-cpu + * (highest priority wins). Done directly because backend-api's KernelServiceLoader is + * jvmMain-only — absent from the library's Android variant — while java.util.ServiceLoader + * works on ART. + * - ":scalar" process → [ScalarKernelProvider] is pinned first; every dispatch site only + * auto-installs when the registry is EMPTY, so the NEON provider never registers and this + * process stays scalar. + * This is what powers the one-phone split-screen race. */ -/** Which kernels the next engine (re)build pins — full-power A/B comparison. */ -enum class KernelMode { NEON, SCALAR } - -class SkainetDemoApp : Application() { +class KernelRaceApp : Application() { private val mutex = Mutex() - private var engine: LlmEngine? = null + private var engine: GenerativeEngine? = null + private val modelProvider by lazy { AndroidModelProvider(this, hfToken = BuildConfig.HF_TOKEN) } val isScalarProcess: Boolean by lazy { currentProcessName().endsWith(":scalar") } @@ -41,13 +47,12 @@ class SkainetDemoApp : Application() { var kernelMode: KernelMode = KernelMode.NEON private set - /** Kernel lookups are cached per execution context, so switching modes - * drops the engine; the next generate reloads with the new pin. */ + /** Kernel lookups are cached per execution context, so switching modes drops the engine; + * the next generate reloads with the new pin. */ suspend fun setKernelMode(mode: KernelMode) = mutex.withLock { if (mode != kernelMode) { kernelMode = mode engine = null - PerfLog.event("kernel_mode_switch", "mode" to mode, "engineDropped" to true) } } @@ -58,42 +63,31 @@ class SkainetDemoApp : Application() { } else { ServiceLoader.load(KernelProvider::class.java).forEach { KernelRegistry.register(it) } } - PerfLog.event( + logEvent( "app_start", "process" to if (isScalarProcess) "scalar" else "main", "providers" to KernelRegistry.availableNames(), ) } - suspend fun engine(onProgress: (String) -> Unit = {}): LlmEngine = mutex.withLock { + suspend fun engine(onProgress: (String) -> Unit = {}): GenerativeEngine = mutex.withLock { engine ?: withContext(Dispatchers.IO) { - val gguf = ModelSource.resolve(this@SkainetDemoApp, onProgress) + val model = modelProvider.resolve(onProgress) onProgress("Loading model…") val kernels = if (isScalarProcess || kernelMode == KernelMode.SCALAR) "scalar" else "neon" - PerfLog.event( - "model_load_start", - "file" to gguf.name, - "fileMB" to gguf.length() / 1_000_000, - "kernels" to kernels, - ) - val loadStart = SystemClock.elapsedRealtime() - LlmEngine.load(gguf).also { + logEvent("model_load_start", "kernels" to kernels) + LlmEngine.load(DirectCpuExecutionContext(), model).also { if (isScalarProcess || kernelMode == KernelMode.SCALAR) { - // Android's platform ops factory re-registers ServiceLoader - // providers (JNI NEON included) on every context creation, - // which defeats the onCreate pin. Kernel lookups are lazy — - // resolved at the FIRST matmul — so re-pin scalar after the - // load has constructed the context, before any forward runs. + // Android's platform ops factory re-registers ServiceLoader providers (JNI + // included) on every context creation, which defeats the onCreate pin. + // Kernel lookups are lazy — resolved at the FIRST matmul — so re-pin scalar + // after the load has constructed the context, before any forward runs. KernelRegistry.clearForTesting() KernelRegistry.register(ScalarKernelProvider) } - // NEON mode needs no action: the load's context creation just - // re-registered the ServiceLoader providers (JNI at priority 100). - PerfLog.event( - "model_load_done", - "durMs" to SystemClock.elapsedRealtime() - loadStart, - "kernels" to kernels, - ) + // NEON mode needs no action: the load's context creation just re-registered + // the ServiceLoader providers (JNI at priority 100). + logEvent("model_load_done", "kernels" to kernels) } }.also { engine = it } } @@ -101,12 +95,11 @@ class SkainetDemoApp : Application() { override fun onTrimMemory(level: Int) { super.onTrimMemory(level) if (level >= ComponentCallbacks2.TRIM_MEMORY_BACKGROUND) { - PerfLog.event("trim_memory", "level" to level, "engineDropped" to (engine != null)) engine = null // reloaded lazily on next generate } } private fun currentProcessName(): String = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.P) getProcessName() - else java.io.File("/proc/self/cmdline").readText().substringBefore('\u0000') + else File("/proc/self/cmdline").readText().substringBefore(0.toChar()) } diff --git a/KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/KernelRaceControls.kt b/KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/KernelRaceControls.kt new file mode 100644 index 0000000..135206e --- /dev/null +++ b/KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/KernelRaceControls.kt @@ -0,0 +1,110 @@ +package sk.ainet.samples.kernelrace + +import android.content.Intent +import androidx.compose.foundation.layout.Row +import androidx.compose.foundation.layout.Spacer +import androidx.compose.foundation.layout.fillMaxWidth +import androidx.compose.foundation.layout.height +import androidx.compose.foundation.layout.width +import androidx.compose.material3.AssistChip +import androidx.compose.material3.FilterChip +import androidx.compose.material3.Text +import androidx.compose.material3.TextButton +import androidx.compose.runtime.Composable +import androidx.compose.runtime.getValue +import androidx.compose.runtime.mutableStateOf +import androidx.compose.runtime.remember +import androidx.compose.runtime.rememberCoroutineScope +import androidx.compose.runtime.saveable.rememberSaveable +import androidx.compose.runtime.setValue +import androidx.compose.ui.Alignment +import androidx.compose.ui.Modifier +import androidx.compose.ui.platform.LocalContext +import androidx.compose.ui.unit.dp +import kotlinx.coroutines.launch +import sk.ainet.samples.kernelrace.platform.kernelTierLabel +import sk.ainet.samples.kernelrace.vm.ChatViewModel + +const val ACTION_RACE = "sk.ainet.samples.kernelrace.action.RACE" +const val EXTRA_PROMPT = "prompt" + +/** + * The Android-only race UI, injected into the platform-neutral [sk.ainet.samples.kernelrace.ui.ChatScreen] + * via its `kernelControls` slot. In the `:scalar` process this collapses to a read-only chip + * (that window is a race participant, not a driver); the main process gets the NEON/SCALAR + * switch plus the split-screen race button. + */ +@Composable +fun KernelRaceControls(busy: Boolean, viewModel: ChatViewModel, currentPrompt: () -> String) { + val context = LocalContext.current + val app = remember { context.applicationContext as KernelRaceApp } + + if (app.isScalarProcess) { + AssistChip(onClick = {}, label = { Text("SCALAR") }) + return + } + + val scope = rememberCoroutineScope() + var scalarMode by rememberSaveable { mutableStateOf(app.kernelMode == KernelMode.SCALAR) } + + // Fullscreen A/B switch: whole phone on one kernel path — record one run per mode and compare. + Row(verticalAlignment = Alignment.CenterVertically) { + FilterChip( + selected = !scalarMode, + onClick = { + scope.launch { + app.setKernelMode(KernelMode.NEON) + scalarMode = false + viewModel.onKernelModeChanged(kernelTierLabel(), scalarMode = false) + } + }, + enabled = !busy, + label = { Text(if (scalarMode) "NEON" else kernelTierLabel()) }, + ) + Spacer(Modifier.width(8.dp)) + FilterChip( + selected = scalarMode, + onClick = { + scope.launch { + app.setKernelMode(KernelMode.SCALAR) + scalarMode = true + viewModel.onKernelModeChanged("SCALAR", scalarMode = true) + } + }, + enabled = !busy, + label = { Text("SCALAR") }, + ) + } + + Spacer(Modifier.height(4.dp)) + + // Saveable: entering split-screen recreates the Activity, and plain remember{} would + // silently disarm the race button. + var raceArmed by rememberSaveable { mutableStateOf(false) } + // Tap 1: open the scalar twin adjacent and preload BOTH engines. + // Tap 2: broadcast the prompt so both windows start in the same instant. + TextButton( + onClick = { + if (!raceArmed) { + context.startActivity( + Intent(context, ScalarActivity::class.java).addFlags( + Intent.FLAG_ACTIVITY_LAUNCH_ADJACENT or Intent.FLAG_ACTIVITY_NEW_TASK + ) + ) + viewModel.preload() + raceArmed = true + } else { + context.sendBroadcast( + Intent(ACTION_RACE) + .setPackage(context.packageName) + .putExtra(EXTRA_PROMPT, currentPrompt()) + ) + viewModel.generate(currentPrompt()) + } + }, + enabled = !busy, + modifier = Modifier.fillMaxWidth(), + ) { + Text(if (raceArmed) "Start race 🏁" else "Race against scalar (split screen)") + } +} diff --git a/KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/MainActivity.kt b/KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/MainActivity.kt new file mode 100644 index 0000000..019a0c1 --- /dev/null +++ b/KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/MainActivity.kt @@ -0,0 +1,33 @@ +package sk.ainet.samples.kernelrace + +import android.os.Bundle +import androidx.activity.ComponentActivity +import androidx.activity.compose.setContent +import androidx.activity.viewModels +import androidx.lifecycle.ViewModel +import androidx.lifecycle.ViewModelProvider +import sk.ainet.samples.kernelrace.vm.ChatViewModel + +class MainActivity : ComponentActivity() { + + private val app get() = applicationContext as KernelRaceApp + + private val viewModel: ChatViewModel by viewModels { + object : ViewModelProvider.Factory { + @Suppress("UNCHECKED_CAST") + override fun create(modelClass: Class): T = + ChatViewModel(loadModel = { onProgress -> app.engine(onProgress) }) as T + } + } + + override fun onCreate(savedInstanceState: Bundle?) { + super.onCreate(savedInstanceState) + setContent { + App( + viewModel = viewModel, + skainetVersion = BuildConfig.SKAINET_VERSION, + kernelControls = { busy, vm, currentPrompt -> KernelRaceControls(busy, vm, currentPrompt) }, + ) + } + } +} diff --git a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ScalarActivity.kt b/KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/ScalarActivity.kt similarity index 51% rename from AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ScalarActivity.kt rename to KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/ScalarActivity.kt index 7da654a..d86e724 100644 --- a/AndroidNeonLlmDemo/app/src/main/kotlin/sk/ainet/demo/ScalarActivity.kt +++ b/KernelRace/composeApp/src/androidMain/kotlin/sk/ainet/samples/kernelrace/ScalarActivity.kt @@ -1,4 +1,4 @@ -package sk.ainet.demo +package sk.ainet.samples.kernelrace import android.content.BroadcastReceiver import android.content.Context @@ -9,20 +9,29 @@ import androidx.activity.ComponentActivity import androidx.activity.compose.setContent import androidx.activity.viewModels import androidx.core.content.ContextCompat -import sk.ainet.ui.theme.SKaiNETTheme +import androidx.lifecycle.ViewModel +import androidx.lifecycle.ViewModelProvider +import sk.ainet.samples.kernelrace.vm.ChatViewModel /** - * The "before" half of the split-screen race. Runs in the `:scalar` - * process, where SkainetDemoApp pins ScalarKernelProvider — same APK, - * same model file, same code, no NEON. + * The "before" half of the split-screen race. Runs in the `:scalar` process, where + * [KernelRaceApp] pins [sk.ainet.exec.kernel.ScalarKernelProvider] — same APK, same model file, + * same code, no NEON. * - * Preloads the engine on launch and starts generating when the main - * window broadcasts [ACTION_RACE], so both sides fire simultaneously - * from a single button. + * Preloads the engine on launch and starts generating when the main window broadcasts + * [ACTION_RACE], so both sides fire simultaneously from a single button. */ class ScalarActivity : ComponentActivity() { - private val viewModel: ChatViewModel by viewModels() + private val app get() = applicationContext as KernelRaceApp + + private val viewModel: ChatViewModel by viewModels { + object : ViewModelProvider.Factory { + @Suppress("UNCHECKED_CAST") + override fun create(modelClass: Class): T = + ChatViewModel(loadModel = { onProgress -> app.engine(onProgress) }) as T + } + } private val raceReceiver = object : BroadcastReceiver() { override fun onReceive(context: Context?, intent: Intent?) { @@ -36,7 +45,13 @@ class ScalarActivity : ComponentActivity() { this, raceReceiver, IntentFilter(ACTION_RACE), ContextCompat.RECEIVER_NOT_EXPORTED, ) viewModel.preload() - setContent { SKaiNETTheme { ChatScreen(viewModel) } } + setContent { + App( + viewModel = viewModel, + skainetVersion = BuildConfig.SKAINET_VERSION, + kernelControls = { busy, vm, currentPrompt -> KernelRaceControls(busy, vm, currentPrompt) }, + ) + } } override fun onDestroy() { diff --git a/AndroidNeonLlmDemo/app/src/main/res/drawable-nodpi/skainet_logo.png b/KernelRace/composeApp/src/commonMain/composeResources/drawable/skainet_logo.png similarity index 100% rename from AndroidNeonLlmDemo/app/src/main/res/drawable-nodpi/skainet_logo.png rename to KernelRace/composeApp/src/commonMain/composeResources/drawable/skainet_logo.png diff --git a/KernelRace/composeApp/src/commonMain/kotlin/sk/ainet/samples/kernelrace/App.kt b/KernelRace/composeApp/src/commonMain/kotlin/sk/ainet/samples/kernelrace/App.kt new file mode 100644 index 0000000..b39c41c --- /dev/null +++ b/KernelRace/composeApp/src/commonMain/kotlin/sk/ainet/samples/kernelrace/App.kt @@ -0,0 +1,17 @@ +package sk.ainet.samples.kernelrace + +import androidx.compose.runtime.Composable +import sk.ainet.samples.kernelrace.ui.ChatScreen +import sk.ainet.samples.kernelrace.vm.ChatViewModel +import sk.ainet.ui.theme.SKaiNETTheme + +@Composable +fun App( + viewModel: ChatViewModel, + skainetVersion: String, + kernelControls: @Composable (busy: Boolean, viewModel: ChatViewModel, currentPrompt: () -> String) -> Unit = { _, _, _ -> }, +) { + SKaiNETTheme { + ChatScreen(viewModel = viewModel, skainetVersion = skainetVersion, kernelControls = kernelControls) + } +} diff --git a/KernelRace/composeApp/src/commonMain/kotlin/sk/ainet/samples/kernelrace/BuildInfo.kt b/KernelRace/composeApp/src/commonMain/kotlin/sk/ainet/samples/kernelrace/BuildInfo.kt new file mode 100644 index 0000000..b273f04 --- /dev/null +++ b/KernelRace/composeApp/src/commonMain/kotlin/sk/ainet/samples/kernelrace/BuildInfo.kt @@ -0,0 +1,5 @@ +package sk.ainet.samples.kernelrace + +/** Desktop/wasm display version — keep in sync with `skainet` in gradle/libs.versions.toml. + * Android reads the real catalog value at build time via `BuildConfig.SKAINET_VERSION` instead. */ +const val SKAINET_VERSION = "0.39.1" diff --git a/KernelRace/composeApp/src/commonMain/kotlin/sk/ainet/samples/kernelrace/ui/ChatScreen.kt b/KernelRace/composeApp/src/commonMain/kotlin/sk/ainet/samples/kernelrace/ui/ChatScreen.kt new file mode 100644 index 0000000..0cdc132 --- /dev/null +++ b/KernelRace/composeApp/src/commonMain/kotlin/sk/ainet/samples/kernelrace/ui/ChatScreen.kt @@ -0,0 +1,138 @@ +package sk.ainet.samples.kernelrace.ui + +import androidx.compose.foundation.Image +import androidx.compose.foundation.layout.Arrangement +import androidx.compose.foundation.layout.Column +import androidx.compose.foundation.layout.Row +import androidx.compose.foundation.layout.Spacer +import androidx.compose.foundation.layout.fillMaxSize +import androidx.compose.foundation.layout.fillMaxWidth +import androidx.compose.foundation.layout.height +import androidx.compose.foundation.layout.padding +import androidx.compose.foundation.layout.size +import androidx.compose.foundation.layout.width +import androidx.compose.foundation.rememberScrollState +import androidx.compose.foundation.verticalScroll +import androidx.compose.material3.Button +import androidx.compose.material3.ExperimentalMaterial3Api +import androidx.compose.material3.MaterialTheme +import androidx.compose.material3.OutlinedTextField +import androidx.compose.material3.Scaffold +import androidx.compose.material3.Text +import androidx.compose.material3.TopAppBar +import androidx.compose.runtime.Composable +import androidx.compose.runtime.collectAsState +import androidx.compose.runtime.getValue +import androidx.compose.runtime.mutableStateOf +import androidx.compose.runtime.remember +import androidx.compose.runtime.setValue +import androidx.compose.ui.Alignment +import androidx.compose.ui.Modifier +import androidx.compose.ui.unit.dp +import org.jetbrains.compose.resources.painterResource +import sk.ainet.samples.kernelrace.vm.ChatViewModel +import sk.ainet.ui.components.indeterminateOrbitingFadingRingLoader +import kernelrace.composeapp.generated.resources.Res +import kernelrace.composeapp.generated.resources.skainet_logo + +@OptIn(ExperimentalMaterial3Api::class) +@Composable +fun ChatScreen( + viewModel: ChatViewModel, + skainetVersion: String, + kernelControls: @Composable (busy: Boolean, viewModel: ChatViewModel, currentPrompt: () -> String) -> Unit, +) { + val state by viewModel.state.collectAsState() + var prompt by remember { mutableStateOf("Explain what a NEON (ARM) instruction is, in two sentences.") } + + Scaffold( + topBar = { + TopAppBar( + title = { + Row(verticalAlignment = Alignment.CenterVertically) { + Image( + painter = painterResource(Res.drawable.skainet_logo), + contentDescription = "SKaiNET logo", + modifier = Modifier.size(36.dp), + ) + Spacer(Modifier.width(10.dp)) + Text("SKaiNET", style = MaterialTheme.typography.titleLarge) + Spacer(Modifier.width(6.dp)) + Text( + skainetVersion, + style = MaterialTheme.typography.labelSmall, + color = MaterialTheme.colorScheme.onSurfaceVariant, + ) + } + }, + actions = { + state.tokensPerSecond?.let { + Text( + "${formatFixed1(it)} tok/s", + style = MaterialTheme.typography.titleLarge, + color = MaterialTheme.colorScheme.primary, + modifier = Modifier.padding(end = 16.dp), + ) + } + }, + ) + }, + ) { padding -> + Column( + Modifier.fillMaxSize().padding(padding) + .padding(horizontal = 16.dp, vertical = 4.dp) + ) { + kernelControls(state.busy, viewModel) { prompt } + + Text( + state.status, + style = MaterialTheme.typography.bodySmall, + maxLines = 2, + modifier = Modifier.fillMaxWidth().padding(top = 2.dp), + ) + Spacer(Modifier.height(6.dp)) + + val loading = state.busy && state.output.isEmpty() + if (loading) { + Column( + Modifier.weight(1f).fillMaxWidth(), + horizontalAlignment = Alignment.CenterHorizontally, + verticalArrangement = Arrangement.Center, + ) { + indeterminateOrbitingFadingRingLoader(size = 120.dp) + Spacer(Modifier.height(16.dp)) + Text(state.status, style = MaterialTheme.typography.bodyMedium) + } + } else { + Text( + text = state.output.ifEmpty { "…" }, + modifier = Modifier.weight(1f).fillMaxWidth().verticalScroll(rememberScrollState()), + style = MaterialTheme.typography.bodyLarge, + ) + } + + Spacer(Modifier.height(8.dp)) + OutlinedTextField( + value = prompt, + onValueChange = { prompt = it }, + modifier = Modifier.fillMaxWidth(), + label = { Text("Prompt") }, + ) + Spacer(Modifier.height(8.dp)) + Button( + onClick = { viewModel.generate(prompt) }, + enabled = !state.busy, + modifier = Modifier.fillMaxWidth(), + ) { + Text(if (state.busy) "Generating…" else "Generate on-device") + } + } + } +} + +private fun formatFixed1(value: Double): String { + val rounded = kotlin.math.round(value * 10) / 10 + val whole = rounded.toLong() + val frac = kotlin.math.round((rounded - whole) * 10).toInt().let { if (it < 0) -it else it } + return "$whole.$frac" +} diff --git a/KernelRace/composeApp/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/main.kt b/KernelRace/composeApp/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/main.kt new file mode 100644 index 0000000..269aa5b --- /dev/null +++ b/KernelRace/composeApp/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/main.kt @@ -0,0 +1,23 @@ +package sk.ainet.samples.kernelrace + +import androidx.compose.runtime.remember +import androidx.compose.ui.window.Window +import androidx.compose.ui.window.application +import sk.ainet.context.DirectCpuExecutionContext +import sk.ainet.samples.kernelrace.engine.LlmEngine +import sk.ainet.samples.kernelrace.model.DesktopModelProvider +import sk.ainet.samples.kernelrace.model.ModelData +import sk.ainet.samples.kernelrace.vm.ChatViewModel + +fun main() = application { + Window(onCloseRequest = ::exitApplication, title = "SKaiNET Kernel Race") { + val viewModel = remember { + ChatViewModel(loadModel = { onProgress -> + val model = DesktopModelProvider().resolve(onProgress) as ModelData.FilePath + onProgress("Building runtime…") + LlmEngine.load(DirectCpuExecutionContext(), model) + }) + } + App(viewModel = viewModel, skainetVersion = SKAINET_VERSION) + } +} diff --git a/KernelRace/composeApp/src/wasmJsMain/kotlin/sk/ainet/samples/kernelrace/main.kt b/KernelRace/composeApp/src/wasmJsMain/kotlin/sk/ainet/samples/kernelrace/main.kt new file mode 100644 index 0000000..7d17410 --- /dev/null +++ b/KernelRace/composeApp/src/wasmJsMain/kotlin/sk/ainet/samples/kernelrace/main.kt @@ -0,0 +1,26 @@ +package sk.ainet.samples.kernelrace + +import androidx.compose.runtime.remember +import androidx.compose.ui.ExperimentalComposeUiApi +import androidx.compose.ui.window.ComposeViewport +import kernelrace.composeapp.generated.resources.Res +import kotlinx.browser.document +import sk.ainet.context.DirectCpuExecutionContext +import sk.ainet.samples.kernelrace.engine.LlmEngine +import sk.ainet.samples.kernelrace.model.ModelData +import sk.ainet.samples.kernelrace.vm.ChatViewModel + +@OptIn(ExperimentalComposeUiApi::class) +fun main() { + ComposeViewport(document.body!!) { + val viewModel = remember { + ChatViewModel(loadModel = { onProgress -> + onProgress("Downloading model bundle…") + val bytes = Res.readBytes("files/SmolLM2-135M-Instruct-Q8_0.gguf") + onProgress("Building runtime…") + LlmEngine.load(DirectCpuExecutionContext(), ModelData.Bytes(bytes)) + }) + } + App(viewModel = viewModel, skainetVersion = SKAINET_VERSION) + } +} diff --git a/KernelRace/composeApp/src/wasmJsMain/resources/app.html b/KernelRace/composeApp/src/wasmJsMain/resources/app.html new file mode 100644 index 0000000..9aa46fc --- /dev/null +++ b/KernelRace/composeApp/src/wasmJsMain/resources/app.html @@ -0,0 +1,12 @@ + + + + + + Kernel Race + + + + + + diff --git a/KernelRace/composeApp/src/wasmJsMain/resources/index.html b/KernelRace/composeApp/src/wasmJsMain/resources/index.html new file mode 100644 index 0000000..b809585 --- /dev/null +++ b/KernelRace/composeApp/src/wasmJsMain/resources/index.html @@ -0,0 +1,216 @@ + + + + + + Kernel Race - SKaiNET Examples + + + + + + +
+ +
+ +
+ +
+ + + +
+ © 2026 SKaiNET. All rights reserved. · + ~145 MB model download on first load — NEON vs scalar race itself is Android-only; this build runs the scalar kernel path. +
+ + + + diff --git a/KernelRace/composeApp/src/wasmJsMain/resources/logo.png b/KernelRace/composeApp/src/wasmJsMain/resources/logo.png new file mode 100644 index 0000000..3ae74ca Binary files /dev/null and b/KernelRace/composeApp/src/wasmJsMain/resources/logo.png differ diff --git a/KernelRace/composeApp/src/wasmJsMain/resources/styles.css b/KernelRace/composeApp/src/wasmJsMain/resources/styles.css new file mode 100644 index 0000000..8e94d43 --- /dev/null +++ b/KernelRace/composeApp/src/wasmJsMain/resources/styles.css @@ -0,0 +1,7 @@ +html, body { + width: 100%; + height: 100%; + margin: 0; + padding: 0; + overflow: hidden; +} diff --git a/AndroidNeonLlmDemo/docs/screenshots/split_race.png b/KernelRace/docs/screenshots/split_race.png similarity index 100% rename from AndroidNeonLlmDemo/docs/screenshots/split_race.png rename to KernelRace/docs/screenshots/split_race.png diff --git a/KernelRace/gradle.properties b/KernelRace/gradle.properties new file mode 100644 index 0000000..a1ad09c --- /dev/null +++ b/KernelRace/gradle.properties @@ -0,0 +1,12 @@ +kotlin.code.style=official + +#Gradle +org.gradle.jvmargs=-Xmx4g -XX:MaxMetaspaceSize=512m -Dfile.encoding=UTF-8 +kotlin.daemon.jvmargs=-Xmx4g + +#Android +android.nonTransitiveRClass=true +android.useAndroidX=true + +#Kotlin Multiplatform +kotlin.mpp.enableCInteropCommonization=true diff --git a/KernelRace/gradle/libs.versions.toml b/KernelRace/gradle/libs.versions.toml new file mode 100644 index 0000000..b1c39d3 --- /dev/null +++ b/KernelRace/gradle/libs.versions.toml @@ -0,0 +1,60 @@ +[versions] +agp = "8.12.3" +android-compileSdk = "36" +android-minSdk = "24" +android-targetSdk = "36" +androidx-activityCompose = "1.12.4" +androidx-lifecycle = "2.9.6" +composeHotReload = "1.0.0" +# Pinned to 1.10.1 to match the shared :skainet-ui design system (included +# build): a mismatched Compose pulls two Skiko versions onto the desktop +# classpath, which crashes with UnsatisfiedLinkError +# (getAdapterMaxTextureSize) at first render on macOS/Metal. +composeMultiplatform = "1.10.1" +# Must match the Kotlin version SKaiNET was built with. +kotlin = "2.4.10" +kotlinx-coroutines = "1.11.0" +kotlinxIo = "0.9.0" +# skainet-transformers has not released against the 0.40.x core line yet +# (tops out at 0.39.1 on Maven Central), so this sample — like KllamaDemo — +# stays one core version behind the transformers-free samples. +skainet = "0.39.1" +skainet-transformers = "0.39.1" + +[libraries] +kotlin-test = { module = "org.jetbrains.kotlin:kotlin-test", version.ref = "kotlin" } +androidx-activity-compose = { module = "androidx.activity:activity-compose", version.ref = "androidx-activityCompose" } +compose-uiTooling = { module = "org.jetbrains.compose.ui:ui-tooling", version.ref = "composeMultiplatform" } +androidx-lifecycle-viewmodel = { module = "org.jetbrains.androidx.lifecycle:lifecycle-viewmodel", version.ref = "androidx-lifecycle" } +androidx-lifecycle-viewmodelCompose = { module = "org.jetbrains.androidx.lifecycle:lifecycle-viewmodel-compose", version.ref = "androidx-lifecycle" } +androidx-lifecycle-runtimeCompose = { module = "org.jetbrains.androidx.lifecycle:lifecycle-runtime-compose", version.ref = "androidx-lifecycle" } +kotlinx-coroutines = { module = "org.jetbrains.kotlinx:kotlinx-coroutines-core", version.ref = "kotlinx-coroutines" } +kotlinx-coroutines-test = { module = "org.jetbrains.kotlinx:kotlinx-coroutines-test", version.ref = "kotlinx-coroutines" } +kotlinx-coroutines-android = { module = "org.jetbrains.kotlinx:kotlinx-coroutines-android", version.ref = "kotlinx-coroutines" } +kotlinx-coroutinesSwing = { module = "org.jetbrains.kotlinx:kotlinx-coroutines-swing", version.ref = "kotlinx-coroutines" } +kotlinx-io-core = { module = "org.jetbrains.kotlinx:kotlinx-io-core", version.ref = "kotlinxIo" } + +# SKaiNET core — BOM-governed +skainet-bom = { module = "sk.ainet:skainet-bom", version.ref = "skainet" } +skainet-lang-core = { module = "sk.ainet.core:skainet-lang-core" } +skainet-backend-cpu = { module = "sk.ainet.core:skainet-backend-cpu" } +# Hand-written ARM NEON kernels — Android/JVM JNI only, not wired into commonMain. +skainet-backend-jni-cpu = { module = "sk.ainet.core:skainet-backend-jni-cpu" } +skainet-data-source = { module = "sk.ainet.core:skainet-data-source" } +skainet-io-core = { module = "sk.ainet.core:skainet-io-core" } +skainet-io-gguf = { module = "sk.ainet.core:skainet-io-gguf" } + +# SKaiNET-transformers — BOM-governed +skainet-transformers-bom = { module = "sk.ainet.transformers:skainet-transformers-bom", version.ref = "skainet-transformers" } +skainet-transformers-core = { module = "sk.ainet.transformers:skainet-transformers-core" } +skainet-transformers-inference-llama = { module = "sk.ainet.transformers:skainet-transformers-inference-llama" } +skainet-transformers-runtime-kllama = { module = "sk.ainet.transformers:skainet-transformers-runtime-kllama" } +skainet-transformers-agent = { module = "sk.ainet.transformers:skainet-transformers-agent" } + +[plugins] +androidApplication = { id = "com.android.application", version.ref = "agp" } +androidLibrary = { id = "com.android.library", version.ref = "agp" } +composeHotReload = { id = "org.jetbrains.compose.hot-reload", version.ref = "composeHotReload" } +composeMultiplatform = { id = "org.jetbrains.compose", version.ref = "composeMultiplatform" } +composeCompiler = { id = "org.jetbrains.kotlin.plugin.compose", version.ref = "kotlin" } +kotlinMultiplatform = { id = "org.jetbrains.kotlin.multiplatform", version.ref = "kotlin" } diff --git a/KernelRace/gradle/wrapper/gradle-wrapper.jar b/KernelRace/gradle/wrapper/gradle-wrapper.jar new file mode 100644 index 0000000..f8e1ee3 Binary files /dev/null and b/KernelRace/gradle/wrapper/gradle-wrapper.jar differ diff --git a/AndroidNeonLlmDemo/gradle/wrapper/gradle-wrapper.properties b/KernelRace/gradle/wrapper/gradle-wrapper.properties similarity index 63% rename from AndroidNeonLlmDemo/gradle/wrapper/gradle-wrapper.properties rename to KernelRace/gradle/wrapper/gradle-wrapper.properties index 8ffd08e..5dd3c01 100644 --- a/AndroidNeonLlmDemo/gradle/wrapper/gradle-wrapper.properties +++ b/KernelRace/gradle/wrapper/gradle-wrapper.properties @@ -1,8 +1,6 @@ -#Tue Aug 11 00:25:18 CEST 2026 distributionBase=GRADLE_USER_HOME distributionPath=wrapper/dists -distributionSha256Sum=553c78f50dafcd54d65b9a444649057857469edf836431389695608536d6b746 -distributionUrl=https\://services.gradle.org/distributions/gradle-9.5.0-bin.zip +distributionUrl=https\://services.gradle.org/distributions/gradle-9.5.1-bin.zip networkTimeout=10000 validateDistributionUrl=true zipStoreBase=GRADLE_USER_HOME diff --git a/AndroidNeonLlmDemo/gradlew b/KernelRace/gradlew similarity index 100% rename from AndroidNeonLlmDemo/gradlew rename to KernelRace/gradlew diff --git a/AndroidNeonLlmDemo/gradlew.bat b/KernelRace/gradlew.bat similarity index 100% rename from AndroidNeonLlmDemo/gradlew.bat rename to KernelRace/gradlew.bat diff --git a/KernelRace/kotlin-js-store/wasm/yarn.lock b/KernelRace/kotlin-js-store/wasm/yarn.lock new file mode 100644 index 0000000..5f4567d --- /dev/null +++ b/KernelRace/kotlin-js-store/wasm/yarn.lock @@ -0,0 +1,8 @@ +# THIS IS AN AUTOGENERATED FILE. DO NOT EDIT THIS FILE DIRECTLY. +# yarn lockfile v1 + + +"@js-joda/core@3.2.0": + version "3.2.0" + resolved "https://registry.yarnpkg.com/@js-joda/core/-/core-3.2.0.tgz#3e61e21b7b2b8a6be746df1335cf91d70db2a273" + integrity sha512-PMqgJ0sw5B7FKb2d5bWYIoxjri+QlW/Pys7+Rw82jSH0QN3rB05jZ/VrrsUdh1w4+i2kw9JOejXGq/KhDOX7Kg== diff --git a/KernelRace/scripts/fetch-model.sh b/KernelRace/scripts/fetch-model.sh new file mode 100755 index 0000000..356e656 --- /dev/null +++ b/KernelRace/scripts/fetch-model.sh @@ -0,0 +1,38 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Fetch the SmolLM2-135M-Instruct GGUF (Q8_0, ~145 MB) into the composeApp's commonMain +# composeResources so the bundled wasmJs / desktop artifacts carry the model too. Android keeps +# its own asset-or-download path (AndroidModelProvider) since a bundled 145 MB asset would bloat +# every install; wasm has no filesystem so this bundled copy is its only option. Q8_0 is what the +# NEON kernels also want on Android, so one file serves every target. Destination is +# .gitignore'd via the repo's *.gguf rule. + +MODEL_URL="https://huggingface.co/unsloth/SmolLM2-135M-Instruct-GGUF/resolve/main/SmolLM2-135M-Instruct-Q8_0.gguf" +DEST_DIR="$(cd "$(dirname "$0")/.." && pwd)/composeApp/src/commonMain/composeResources/files" +DEST_FILE="$DEST_DIR/SmolLM2-135M-Instruct-Q8_0.gguf" +MIN_SIZE_BYTES=$((100 * 1024 * 1024)) # sanity floor: 100 MB (Q8_0 is ~145 MB) + +mkdir -p "$DEST_DIR" + +if [[ -f "$DEST_FILE" ]]; then + size=$(stat -f%z "$DEST_FILE" 2>/dev/null || stat -c%s "$DEST_FILE") + if (( size > MIN_SIZE_BYTES )); then + echo "Model already present at $DEST_FILE ($(( size / 1024 / 1024 )) MB) — skipping." + exit 0 + fi + echo "Existing $DEST_FILE looks truncated ($(( size / 1024 / 1024 )) MB) — re-downloading." + rm -f "$DEST_FILE" +fi + +echo "Fetching SmolLM2-135M-Instruct-Q8_0.gguf (~145 MB) from Hugging Face..." +curl -L --fail --progress-bar -o "$DEST_FILE" "$MODEL_URL" + +size=$(stat -f%z "$DEST_FILE" 2>/dev/null || stat -c%s "$DEST_FILE") +if (( size < MIN_SIZE_BYTES )); then + echo "Downloaded file is suspiciously small ($(( size / 1024 / 1024 )) MB). Aborting." >&2 + rm -f "$DEST_FILE" + exit 1 +fi + +echo "Model saved to $DEST_FILE ($(( size / 1024 / 1024 )) MB)." diff --git a/AndroidNeonLlmDemo/settings.gradle.kts b/KernelRace/settings.gradle.kts similarity index 67% rename from AndroidNeonLlmDemo/settings.gradle.kts rename to KernelRace/settings.gradle.kts index fa99a9b..f791703 100644 --- a/AndroidNeonLlmDemo/settings.gradle.kts +++ b/KernelRace/settings.gradle.kts @@ -1,4 +1,5 @@ -rootProject.name = "AndroidNeonLlmDemo" +rootProject.name = "KernelRace" +enableFeaturePreview("TYPESAFE_PROJECT_ACCESSORS") pluginManagement { repositories { @@ -24,15 +25,13 @@ dependencyResolutionManagement { } } mavenCentral() - mavenLocal { - mavenContent { - includeGroupAndSubgroups("sk.ainet") - } - } } } -// Shared SKaiNET design system (theme, logo colors, FadingRingLoader) +// Shared SKaiNET design system (theme + components). Consumed as an included +// build so the example always uses the local source, matching the sibling +// KllamaDemo / TinyTransformer examples. includeBuild("../skainet-ui") -include(":app") +include(":composeApp") +include(":shared") diff --git a/KernelRace/shared/build.gradle.kts b/KernelRace/shared/build.gradle.kts new file mode 100644 index 0000000..0dc217d --- /dev/null +++ b/KernelRace/shared/build.gradle.kts @@ -0,0 +1,83 @@ +import org.jetbrains.kotlin.gradle.ExperimentalWasmDsl + +plugins { + alias(libs.plugins.kotlinMultiplatform) + alias(libs.plugins.androidLibrary) +} + +kotlin { + jvmToolchain(21) + + androidTarget() + + jvm() + + @OptIn(ExperimentalWasmDsl::class) + wasmJs { + browser() + } + + sourceSets { + commonMain.dependencies { + implementation(libs.kotlinx.coroutines) + implementation(libs.kotlinx.io.core) + implementation(libs.androidx.lifecycle.viewmodel) + + // SKaiNET core. BOM applied via api() so its version constraints also reach + // composeApp's classpath — implementation-scoped constraints stay internal to + // this module and leave downstream consumers with unversioned coordinates. + api(project.dependencies.platform(libs.skainet.bom)) + // api, not implementation: composeApp constructs DirectCpuExecutionContext (and + // reads ExecutionContext/FP32/etc.) directly, so these need to be on its classpath. + api(libs.skainet.lang.core) + api(libs.skainet.backend.cpu) + api(libs.skainet.io.core) + api(libs.skainet.io.gguf) + + // SKaiNET-transformers: Llama inference + generation. agent and + // runtime-kllama both publish wasmJs targets as of 0.39.1, so the + // whole engine (unlike KllamaDemo's 0.34.1-era jvmMain-only split) + // lives in commonMain. + api(project.dependencies.platform(libs.skainet.transformers.bom)) + api(libs.skainet.transformers.inference.llama) + api(libs.skainet.transformers.core) + api(libs.skainet.transformers.runtime.kllama) + api(libs.skainet.transformers.agent) + } + androidMain.dependencies { + // hf:// / https:// model download from the Hugging Face Hub + implementation(libs.skainet.data.source) + // Hand-written ARM NEON kernels, dispatched at runtime (armv8-a / armv8.2-a+fp16+dotprod) + implementation(libs.skainet.backend.jni.cpu) + } + jvmMain.dependencies { + implementation(libs.skainet.data.source) + } + commonTest.dependencies { + implementation(libs.kotlin.test) + implementation(libs.kotlinx.coroutines.test) + } + } +} + +tasks.withType().configureEach { + // SIMD-accelerated CPU ops via JDK Vector API (incubator). Same flags the + // composeApp desktop run uses, so jvmTest exercises the same code path. + jvmArgs( + "--add-modules", "jdk.incubator.vector", + "--enable-preview", + "-Dskainet.cpu.vector.enabled=true", + ) +} + +android { + namespace = "sk.ainet.samples.kernelrace.shared" + compileSdk = libs.versions.android.compileSdk.get().toInt() + compileOptions { + sourceCompatibility = JavaVersion.VERSION_21 + targetCompatibility = JavaVersion.VERSION_21 + } + defaultConfig { + minSdk = libs.versions.android.minSdk.get().toInt() + } +} diff --git a/KernelRace/shared/src/androidMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.android.kt b/KernelRace/shared/src/androidMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.android.kt new file mode 100644 index 0000000..6a20fe1 --- /dev/null +++ b/KernelRace/shared/src/androidMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.android.kt @@ -0,0 +1,36 @@ +package sk.ainet.samples.kernelrace.engine + +import sk.ainet.apps.llm.OptimizedLLMMode +import sk.ainet.apps.llm.OptimizedLLMRuntime +import sk.ainet.apps.llm.tokenizer.TokenizerFactory +import sk.ainet.context.ExecutionContext +import sk.ainet.io.AndroidRandomAccessSource +import sk.ainet.io.model.QuantPolicy +import sk.ainet.lang.types.FP32 +import sk.ainet.models.llama.DecoderGgufWeightLoader +import sk.ainet.models.llama.LlamaNetworkLoader +import sk.ainet.samples.kernelrace.model.ModelData + +/** + * The "five lines" fast path: random-access reads straight off the file, + * `QuantPolicy.NATIVE_OPTIMIZED` keeps Q8_0 packed for the hand-written ARM NEON kernels + * (skainet-backend-jni-cpu), which register themselves through the kernel SPI just by being + * on the classpath. + */ +actual suspend fun buildLlamaComponents(ctx: ExecutionContext, model: ModelData): LlamaComponents { + val path = (model as ModelData.FilePath).path + val weights = DecoderGgufWeightLoader( + randomAccessProvider = { AndroidRandomAccessSource.open(path) }, + quantPolicy = QuantPolicy.NATIVE_OPTIMIZED, + acceptedArchitectures = setOf("llama", "mistral"), // SmolLM2 is llama-family + ).loadToMapStreaming(ctx) + val runtime = OptimizedLLMRuntime( + model = LlamaNetworkLoader.fromWeights(weights), + ctx = ctx, + mode = OptimizedLLMMode.DIRECT, + dtype = FP32::class, + bos = weights.metadata.bosTokenId, + ) + val tokenizer = AndroidRandomAccessSource.open(path).use { TokenizerFactory.fromGgufSource(it) } + return LlamaComponents(runtime, tokenizer) +} diff --git a/KernelRace/shared/src/androidMain/kotlin/sk/ainet/samples/kernelrace/model/AndroidModelProvider.kt b/KernelRace/shared/src/androidMain/kotlin/sk/ainet/samples/kernelrace/model/AndroidModelProvider.kt new file mode 100644 index 0000000..efb8f6c --- /dev/null +++ b/KernelRace/shared/src/androidMain/kotlin/sk/ainet/samples/kernelrace/model/AndroidModelProvider.kt @@ -0,0 +1,114 @@ +package sk.ainet.samples.kernelrace.model + +import android.content.Context +import java.io.File +import java.io.FileNotFoundException +import kotlinx.io.Buffer +import kotlinx.io.buffered +import kotlinx.io.files.SystemFileSystem +import kotlin.time.TimeSource +import sk.ainet.data.source.KtorRemoteDataSourceFetcher +import sk.ainet.samples.kernelrace.platform.logEvent +import kotlinx.io.files.Path as KotlinxPath + +/** + * Resolves the GGUF file: bundled asset if present (offline demo builds), otherwise streamed + * from the Hugging Face Hub via SKaiNET's Ktor fetcher into filesDir. Chunked copy straight to + * disk — the model never sits on the ART heap, and progress is reported per real received bytes. + * (Not the resolver's CachePolicy.Bypass path: that buffers the whole artifact in memory before + * returning — exactly what a 145 MB file on a 256 MB heap must avoid.) + * + * Both app processes (NEON and :scalar) share filesDir, so the model is fetched once per device. + */ +class AndroidModelProvider( + private val context: Context, + private val hfToken: String = "", +) : ModelProvider { + + override suspend fun resolve(onProgress: (String) -> Unit): ModelData { + val dir = File(context.filesDir, "models").apply { mkdirs() } + val target = File(dir, HF_FILE) + + return when (val plan = ModelResolver.plan(target.path, target.exists(), assetAvailable(context))) { + is ResolutionPlan.UseCached -> { + logEvent("model_cached", "file" to target.name, "fileMB" to target.length() / 1_000_000) + ModelData.FilePath(plan.path) + } + is ResolutionPlan.UseAsset -> { + copyFromAssets(context, target) + logEvent("model_from_assets", "file" to target.name, "fileMB" to target.length() / 1_000_000) + ModelData.FilePath(plan.path) + } + ResolutionPlan.Download -> { + download(target, onProgress) + ModelData.FilePath(target.path) + } + } + } + + private fun assetAvailable(context: Context): Boolean = try { + context.assets.open(HF_FILE).close() + true + } catch (_: FileNotFoundException) { + false + } + + private fun copyFromAssets(context: Context, target: File) { + context.assets.open(HF_FILE).use { input -> + target.outputStream().use { output -> input.copyTo(output) } + } + } + + private suspend fun download(target: File, onProgress: (String) -> Unit) { + val url = "https://huggingface.co/$HF_REPO/resolve/main/$HF_FILE" + val headers = hfToken.takeIf { it.isNotBlank() }?.let { mapOf("Authorization" to "Bearer $it") } ?: emptyMap() + + val tmp = File(target.parentFile, ModelResolver.partPath(target.name)) + val fetcher = KtorRemoteDataSourceFetcher() + logEvent("download_start", "url" to url) + val startedAt = TimeSource.Monotonic.markNow() + var received = 0L + try { + val content = fetcher.fetch(url, headers) + val totalMb = content.sizeBytes?.let { formatWholeMb(it.toDouble()) } ?: "?" + content.source.use { source -> + SystemFileSystem.sink(KotlinxPath(tmp.path)).buffered().use { sink -> + val chunk = Buffer() + var lastReported = -1L + while (true) { + val n = source.readAtMostTo(chunk, 1024 * 1024) + if (n == -1L) break + sink.write(chunk, n) + received += n + val mb = received / 1_000_000 + if (mb != lastReported) { + lastReported = mb + onProgress("Downloading model… $mb / $totalMb MB") + } + } + } + } + check(tmp.renameTo(target)) { "rename failed: $tmp -> $target" } + val durMs = startedAt.elapsedNow().inWholeMilliseconds + logEvent( + "download_done", + "fileMB" to received / 1_000_000, + "durMs" to durMs, + "MBps" to if (durMs > 0) formatWholeMb(received / (durMs / 1000.0)) else null, + ) + } catch (e: Exception) { + tmp.delete() + logEvent( + "download_failed", + "receivedMB" to received / 1_000_000, + "durMs" to startedAt.elapsedNow().inWholeMilliseconds, + "error" to (e.message ?: e::class.simpleName), + ) + throw e + } finally { + fetcher.close() + } + } +} + +private fun formatWholeMb(bytes: Double): String = (bytes / 1e6).toLong().toString() diff --git a/KernelRace/shared/src/androidMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.android.kt b/KernelRace/shared/src/androidMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.android.kt new file mode 100644 index 0000000..c8e28fc --- /dev/null +++ b/KernelRace/shared/src/androidMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.android.kt @@ -0,0 +1,28 @@ +package sk.ainet.samples.kernelrace.platform + +import android.util.Log +import java.io.File +import sk.ainet.exec.kernel.jni.JniKernelProvider + +private const val NUL_CHAR = 0.toChar() + +/** + * NEON tier label from hardware capability (dispatched .so tier), NOT from KernelRegistry — + * the registry is mutated by mode switches (a scalar-mode run leaves it as scalar until the + * next load), which would mislabel the NEON chip "SCALAR" after switching back. + */ +actual fun kernelTierLabel(): String = + if (JniKernelProvider.isAvailable()) "ARM NEON" else "NEON (unavailable)" + +actual val supportsKernelRace: Boolean = true + +/** ":scalar" vs main process, read once from /proc/self/cmdline — splits log tags so the + * one-phone NEON-vs-scalar race can be watched as two separately filterable `adb logcat` streams. */ +private val processTag: String by lazy { + val cmdline = File("/proc/self/cmdline").readText().substringBefore(NUL_CHAR) + if (cmdline.endsWith(":scalar")) "SCALAR" else "NEON" +} + +internal actual fun platformLog(line: String) { + Log.i("SKAINET_PERF_$processTag", line) +} diff --git a/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/domain/ChatModels.kt b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/domain/ChatModels.kt new file mode 100644 index 0000000..e8494ee --- /dev/null +++ b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/domain/ChatModels.kt @@ -0,0 +1,10 @@ +package sk.ainet.samples.kernelrace.domain + +data class UiState( + val status: String = "Model not loaded", + val kernelTier: String = "", + val output: String = "", + val tokensPerSecond: Double? = null, + val busy: Boolean = false, + val scalarMode: Boolean = false, +) diff --git a/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/domain/TokenStats.kt b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/domain/TokenStats.kt new file mode 100644 index 0000000..d5d13fa --- /dev/null +++ b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/domain/TokenStats.kt @@ -0,0 +1,31 @@ +package sk.ainet.samples.kernelrace.domain + +/** + * Running decode-speed estimate from streamed token arrival times. + * + * Excludes prefill: the clock starts at the *first* emitted token, not at + * generation start, so prompt-processing time never pollutes the tok/s + * number. Also excludes the first token itself from the count (there is no + * decode interval before it), and requires at least half a second of signal + * before reporting a rate — otherwise a couple of fast tokens produce a + * wildly noisy estimate. + */ +class TokenStats { + private var firstTokenAtMs: Long = -1 + var tokenCount: Int = 0 + private set + + /** Call once per emitted token with the current elapsed time (any monotonic clock, ms). */ + fun onToken(nowMs: Long) { + if (tokenCount == 0) firstTokenAtMs = nowMs + tokenCount++ + } + + /** Decode tokens/sec so far, or null if there isn't enough signal yet. */ + fun tokensPerSecond(nowMs: Long): Double? { + if (tokenCount == 0 || firstTokenAtMs < 0) return null + val elapsedS = (nowMs - firstTokenAtMs) / 1000.0 + if (elapsedS <= 0.5) return null + return (tokenCount - 1) / elapsedS + } +} diff --git a/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.kt b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.kt new file mode 100644 index 0000000..470ef3c --- /dev/null +++ b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.kt @@ -0,0 +1,44 @@ +package sk.ainet.samples.kernelrace.engine + +import kotlinx.io.Buffer +import sk.ainet.apps.llm.OptimizedLLMMode +import sk.ainet.apps.llm.OptimizedLLMRuntime +import sk.ainet.apps.llm.Tokenizer +import sk.ainet.apps.llm.tokenizer.GGUFTokenizer +import sk.ainet.context.ExecutionContext +import sk.ainet.io.model.QuantPolicy +import sk.ainet.lang.types.FP32 +import sk.ainet.models.llama.LlamaNetworkLoader +import sk.ainet.samples.kernelrace.model.ModelData + +/** Runtime + tokenizer built from the same GGUF bytes — kept together since both come from one load. */ +class LlamaComponents(val runtime: OptimizedLLMRuntime, val tokenizer: Tokenizer) + +/** + * Platform-split loader: + * - Android takes the "five lines" fast path: random-access reads straight off the file, + * `QuantPolicy.NATIVE_OPTIMIZED` keeps Q8_0 packed for the hand-written ARM NEON kernels. + * - Desktop JVM takes the same file-based path (no NEON kernels to dispatch to on non-Android + * hardware, but still avoids materializing the whole file on-heap). + * - wasmJs has no filesystem, so it falls back to [buildLlamaComponentsFallback]: a sequential + * in-memory `Buffer` + `QuantPolicy.DEQUANTIZE_TO_FP32`. Same correctness, much slower — + * the packed-quant kernels are JVM/Android-only. + */ +expect suspend fun buildLlamaComponents(ctx: ExecutionContext, model: ModelData): LlamaComponents + +/** Multiplatform fallback used by wasmJs (and available to every platform for testing). */ +suspend fun buildLlamaComponentsFallback(ctx: ExecutionContext, bytes: ByteArray): LlamaComponents { + val tokenizer = GGUFTokenizer.fromSource(Buffer().apply { write(bytes) }, false) + val sourceProvider = { Buffer().apply { write(bytes) } } + val model = LlamaNetworkLoader + .fromGguf(sourceProvider, QuantPolicy.DEQUANTIZE_TO_FP32, false) + .load(ctx) + val runtime = OptimizedLLMRuntime( + model = model, + ctx = ctx, + mode = OptimizedLLMMode.DIRECT, + dtype = FP32::class, + bos = tokenizer.bosTokenId, + ) + return LlamaComponents(runtime, tokenizer) +} diff --git a/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/engine/LlmEngine.kt b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/engine/LlmEngine.kt new file mode 100644 index 0000000..89774df --- /dev/null +++ b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/engine/LlmEngine.kt @@ -0,0 +1,81 @@ +package sk.ainet.samples.kernelrace.engine + +import kotlin.math.round +import kotlin.time.TimeMark +import kotlin.time.TimeSource +import sk.ainet.apps.kllama.agent.generateUntilStop +import sk.ainet.context.ExecutionContext +import sk.ainet.samples.kernelrace.model.ModelData +import sk.ainet.samples.kernelrace.platform.logEvent + +/** What [sk.ainet.samples.kernelrace.vm.ChatViewModel] needs from an engine — lets tests + * substitute a fake without touching SKaiNET's own (private-constructor) [LlmEngine]. */ +interface GenerativeEngine { + suspend fun generate(prompt: String, maxTokens: Int = 200, onToken: (String) -> Unit): String +} + +/** + * Loads a GGUF LLM and streams generated tokens — the whole integration. + * + * On Android this is the Scene 4 shot of the demo video: GGUF in, tokens out, no Python, + * no C++ in the app build. The hand-written ARM NEON kernels (skainet-backend-jni-cpu) + * register themselves through the kernel SPI just by being on the classpath. + */ +class LlmEngine private constructor(private val components: LlamaComponents) : GenerativeEngine { + + private val tokenizer get() = components.tokenizer + private val runtime get() = components.runtime + + override suspend fun generate(prompt: String, maxTokens: Int, onToken: (String) -> Unit): String { + runtime.reset() + val templated = chatMlEnvelope(prompt) + val promptTokens = tokenizer.encode(templated) + logEvent( + "generate_start", + "promptTokens" to promptTokens.size, + "maxTokens" to maxTokens, + "eos" to tokenizer.eosTokenId, + ) + val genStart = TimeSource.Monotonic.markNow() + var firstTokenAt: TimeMark? = null + var tokenCount = 0 + val result = runtime.generateUntilStop( + prompt = promptTokens, + maxTokens = maxTokens, + eosTokenId = tokenizer.eosTokenId, + temperature = 0.7f, + onToken = { tokenId -> + if (tokenCount == 0) firstTokenAt = TimeSource.Monotonic.markNow() + tokenCount++ + onToken(tokenizer.decode(tokenId)) + }, + decode = { tokenizer.decode(it) }, + ) + val totalMs = genStart.elapsedNow().inWholeMilliseconds + val decodeS = firstTokenAt?.elapsedNow()?.inWholeMilliseconds?.let { it / 1000.0 } ?: 0.0 + logEvent( + "generate_done", + "tokens" to tokenCount, + "totalMs" to totalMs, + "decodeTokPerSec" to if (tokenCount > 1 && decodeS > 0) { + formatFixed2((tokenCount - 1) / decodeS) + } else null, + "textLen" to result.text.length, + ) + return result.text + } + + companion object { + /** Call off the main thread — streams weights, never materializes the whole file on-heap. */ + suspend fun load(ctx: ExecutionContext, model: ModelData): LlmEngine = + LlmEngine(buildLlamaComponents(ctx, model)) + } +} + +/** `"%.2f".format` isn't available in common code — round to 2 decimals by hand. */ +private fun formatFixed2(value: Double): String { + val rounded = round(value * 100) / 100 + val whole = rounded.toLong() + val frac = round((rounded - whole) * 100).toInt().let { if (it < 0) -it else it } + return "$whole.${frac.toString().padStart(2, '0')}" +} diff --git a/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/engine/PromptTemplate.kt b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/engine/PromptTemplate.kt new file mode 100644 index 0000000..9f4d0ca --- /dev/null +++ b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/engine/PromptTemplate.kt @@ -0,0 +1,8 @@ +package sk.ainet.samples.kernelrace.engine + +/** + * SmolLM2-Instruct is a ChatML model — a raw prompt makes it emit + * `<|im_end|>` immediately. Wrap it in the ChatML envelope it was trained on. + */ +fun chatMlEnvelope(prompt: String): String = + "<|im_start|>user\n$prompt<|im_end|>\n<|im_start|>assistant\n" diff --git a/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/model/ModelData.kt b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/model/ModelData.kt new file mode 100644 index 0000000..6711bfe --- /dev/null +++ b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/model/ModelData.kt @@ -0,0 +1,14 @@ +package sk.ainet.samples.kernelrace.model + +/** + * Where the GGUF weights ended up. [FilePath] platforms (Android, desktop) get random-access + * reads straight off disk; [Bytes] platforms (wasm — no filesystem) hold the whole model + * in memory. + */ +sealed interface ModelData { + data class FilePath(val path: String) : ModelData + data class Bytes(val bytes: ByteArray) : ModelData +} + +const val HF_REPO = "unsloth/SmolLM2-135M-Instruct-GGUF" +const val HF_FILE = "SmolLM2-135M-Instruct-Q8_0.gguf" diff --git a/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/model/ModelProvider.kt b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/model/ModelProvider.kt new file mode 100644 index 0000000..96326c8 --- /dev/null +++ b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/model/ModelProvider.kt @@ -0,0 +1,11 @@ +package sk.ainet.samples.kernelrace.model + +/** + * Resolves the GGUF onto this platform. Android and desktop implementations + * ([AndroidModelProvider], [DesktopModelProvider]) do real file IO (cache hit → bundled asset → + * streamed Hugging Face download); wasm has no filesystem, so composeApp reads the bundled + * Compose resource bytes directly and wraps them as [ModelData.Bytes] — no provider needed there. + */ +interface ModelProvider { + suspend fun resolve(onProgress: (String) -> Unit): ModelData +} diff --git a/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/model/ModelResolver.kt b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/model/ModelResolver.kt new file mode 100644 index 0000000..384ab93 --- /dev/null +++ b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/model/ModelResolver.kt @@ -0,0 +1,27 @@ +package sk.ainet.samples.kernelrace.model + +/** What to do to get the GGUF onto local storage, decided without touching any IO. */ +sealed interface ResolutionPlan { + data class UseCached(val path: String) : ResolutionPlan + data class UseAsset(val path: String) : ResolutionPlan + data object Download : ResolutionPlan +} + +/** + * Pure resolution policy, ported from the original Android-only `ModelSource.resolve`: + * cache hit short-circuits, otherwise a bundled offline asset wins over downloading. + * Takes plain booleans/paths so it needs no filesystem access — the file-based platform + * providers ([AndroidModelProvider], [DesktopModelProvider]) do the actual IO and defer the + * decision here. + */ +object ModelResolver { + fun plan(targetPath: String, targetExists: Boolean, assetAvailable: Boolean): ResolutionPlan = when { + targetExists -> ResolutionPlan.UseCached(targetPath) + assetAvailable -> ResolutionPlan.UseAsset(targetPath) + else -> ResolutionPlan.Download + } + + /** Streamed downloads land here first; an interrupted download leaves this behind, + * never a truncated [targetPath], so a retry doesn't mistake it for a finished file. */ + fun partPath(targetPath: String): String = "$targetPath.part" +} diff --git a/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.kt b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.kt new file mode 100644 index 0000000..5b63003 --- /dev/null +++ b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.kt @@ -0,0 +1,30 @@ +package sk.ainet.samples.kernelrace.platform + +import kotlin.time.TimeSource + +/** Human-readable label for the kernel path this platform/process is running — shown in the UI. */ +expect fun kernelTierLabel(): String + +/** Only Android can pin a JNI NEON provider vs a scalar one and race them side by side. */ +expect val supportsKernelRace: Boolean + +/** Where the platform actually writes a log line — `adb logcat` on Android, stdout elsewhere. */ +internal expect fun platformLog(line: String) + +private val processStart = TimeSource.Monotonic.markNow() + +/** + * One-line structured perf events, greppable with `adb logcat -s SKAINET_PERF_NEON:I` on + * Android (see the platform actual for the process-tagged variant), or read straight off + * stdout on desktop/wasm. + */ +fun logEvent(name: String, vararg details: Pair) { + val line = buildString { + append("event=").append(name) + append(" | +").append(processStart.elapsedNow().inWholeMilliseconds).append("ms") + for ((key, value) in details) { + if (value != null) append(" | ").append(key).append('=').append(value) + } + } + platformLog(line) +} diff --git a/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/vm/ChatViewModel.kt b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/vm/ChatViewModel.kt new file mode 100644 index 0000000..626f887 --- /dev/null +++ b/KernelRace/shared/src/commonMain/kotlin/sk/ainet/samples/kernelrace/vm/ChatViewModel.kt @@ -0,0 +1,98 @@ +package sk.ainet.samples.kernelrace.vm + +import androidx.lifecycle.ViewModel +import androidx.lifecycle.viewModelScope +import kotlin.time.TimeSource +import kotlinx.coroutines.CoroutineDispatcher +import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.flow.MutableStateFlow +import kotlinx.coroutines.flow.StateFlow +import kotlinx.coroutines.launch +import kotlinx.coroutines.withContext +import sk.ainet.samples.kernelrace.domain.TokenStats +import sk.ainet.samples.kernelrace.domain.UiState +import sk.ainet.samples.kernelrace.engine.GenerativeEngine +import sk.ainet.samples.kernelrace.platform.kernelTierLabel + +/** + * Drives model loading + generation. Holds no Android/JVM/wasm-specific state: how the model + * bytes are resolved (Android assets/download, desktop cache, bundled wasm bytes) and how the + * execution context + kernel pinning are set up is entirely the caller's job, injected as a + * suspend lambda — which is what makes this class unit-testable with a fake on every target. + * + * [generationDispatcher] defaults to [Dispatchers.Default] (real CPU-bound work off the UI + * thread) but is overridable so tests can run generation on the same virtual-time scheduler + * as [androidx.lifecycle.viewModelScope]'s Main dispatcher. + */ +class ChatViewModel( + private val loadModel: suspend (onProgress: (String) -> Unit) -> GenerativeEngine, + initialKernelTier: String = kernelTierLabel(), + private val generationDispatcher: CoroutineDispatcher = Dispatchers.Default, +) : ViewModel() { + + private val _state = MutableStateFlow(UiState(kernelTier = initialKernelTier)) + val state: StateFlow = _state + + private var engine: GenerativeEngine? = null + + fun generate(prompt: String) { + if (_state.value.busy || prompt.isBlank()) return + _state.value = _state.value.copy(busy = true, output = "", tokensPerSecond = null, status = "Loading model…") + + viewModelScope.launch { + try { + val loaded = engine ?: loadModel { progress -> + _state.value = _state.value.copy(status = progress) + }.also { engine = it } + _state.value = _state.value.copy(status = "Generating") + + val stats = TokenStats() + val clock = TimeSource.Monotonic.markNow() + withContext(generationDispatcher) { + loaded.generate(prompt) { piece -> + val nowMs = clock.elapsedNow().inWholeMilliseconds + stats.onToken(nowMs) + _state.value = _state.value.copy( + output = _state.value.output + piece, + tokensPerSecond = stats.tokensPerSecond(nowMs), + ) + } + } + _state.value = _state.value.copy( + busy = false, + status = "Done — ${stats.tokenCount} tokens, fully on-device", + ) + } catch (e: Exception) { + _state.value = _state.value.copy(busy = false, status = "Error: ${e.message ?: e::class.simpleName}") + } + } + } + + /** Loads the engine without generating — arms a fair race (Android split-screen only). */ + fun preload() { + if (_state.value.busy) return + _state.value = _state.value.copy(busy = true) + viewModelScope.launch { + try { + engine = loadModel { progress -> _state.value = _state.value.copy(status = progress) } + _state.value = _state.value.copy(busy = false, status = "Model loaded — ready to race") + } catch (e: Exception) { + _state.value = _state.value.copy(busy = false, status = "Error: ${e.message ?: e::class.simpleName}") + } + } + } + + /** Drops the cached engine so the next generate reloads under a different kernel pin + * (Android's fullscreen NEON/SCALAR switch calls this after re-pinning the registry). */ + fun onKernelModeChanged(newKernelTier: String, scalarMode: Boolean) { + if (_state.value.busy) return + engine = null + _state.value = _state.value.copy( + scalarMode = scalarMode, + kernelTier = newKernelTier, + output = "", + tokensPerSecond = null, + status = "Kernel mode changed — model reloads on next run", + ) + } +} diff --git a/KernelRace/shared/src/commonTest/kotlin/sk/ainet/samples/kernelrace/domain/TokenStatsTest.kt b/KernelRace/shared/src/commonTest/kotlin/sk/ainet/samples/kernelrace/domain/TokenStatsTest.kt new file mode 100644 index 0000000..581137b --- /dev/null +++ b/KernelRace/shared/src/commonTest/kotlin/sk/ainet/samples/kernelrace/domain/TokenStatsTest.kt @@ -0,0 +1,45 @@ +package sk.ainet.samples.kernelrace.domain + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertNull + +class TokenStatsTest { + + @Test + fun noRateBeforeAnyToken() { + val stats = TokenStats() + + assertNull(stats.tokensPerSecond(nowMs = 0)) + assertEquals(0, stats.tokenCount) + } + + @Test + fun noRateWithinFirstHalfSecond() { + val stats = TokenStats() + + stats.onToken(nowMs = 0) + stats.onToken(nowMs = 200) + stats.onToken(nowMs = 400) + + // elapsed since the first token is only 400ms — below the 0.5s noise guard + assertNull(stats.tokensPerSecond(nowMs = 400)) + } + + @Test + fun excludesPrefillAndFirstTokenFromTheRate() { + val stats = TokenStats() + + // "prefill" — arrives long before the first decoded token — must not skew the clock + stats.onToken(nowMs = 5_000) + // 4 more tokens over exactly 1 second after the first + stats.onToken(nowMs = 5_250) + stats.onToken(nowMs = 5_500) + stats.onToken(nowMs = 5_750) + stats.onToken(nowMs = 6_000) + + // 5 tokens total, but the rate is measured over 4 decode intervals in 1s => 4 tok/s + assertEquals(5, stats.tokenCount) + assertEquals(4.0, stats.tokensPerSecond(nowMs = 6_000)) + } +} diff --git a/KernelRace/shared/src/commonTest/kotlin/sk/ainet/samples/kernelrace/engine/PromptTemplateTest.kt b/KernelRace/shared/src/commonTest/kotlin/sk/ainet/samples/kernelrace/engine/PromptTemplateTest.kt new file mode 100644 index 0000000..95ebb1c --- /dev/null +++ b/KernelRace/shared/src/commonTest/kotlin/sk/ainet/samples/kernelrace/engine/PromptTemplateTest.kt @@ -0,0 +1,27 @@ +package sk.ainet.samples.kernelrace.engine + +import kotlin.test.Test +import kotlin.test.assertEquals + +class PromptTemplateTest { + + @Test + fun wrapsPromptInChatMlEnvelope() { + val envelope = chatMlEnvelope("Explain NEON in two sentences.") + + assertEquals( + "<|im_start|>user\nExplain NEON in two sentences.<|im_end|>\n<|im_start|>assistant\n", + envelope, + ) + } + + @Test + fun preservesEmbeddedNewlines() { + val envelope = chatMlEnvelope("line one\nline two") + + assertEquals( + "<|im_start|>user\nline one\nline two<|im_end|>\n<|im_start|>assistant\n", + envelope, + ) + } +} diff --git a/KernelRace/shared/src/commonTest/kotlin/sk/ainet/samples/kernelrace/model/ModelResolverTest.kt b/KernelRace/shared/src/commonTest/kotlin/sk/ainet/samples/kernelrace/model/ModelResolverTest.kt new file mode 100644 index 0000000..da90c79 --- /dev/null +++ b/KernelRace/shared/src/commonTest/kotlin/sk/ainet/samples/kernelrace/model/ModelResolverTest.kt @@ -0,0 +1,36 @@ +package sk.ainet.samples.kernelrace.model + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertIs + +class ModelResolverTest { + + @Test + fun cacheHitShortCircuitsBeforeCheckingTheAsset() { + val plan = ModelResolver.plan(targetPath = "/models/x.gguf", targetExists = true, assetAvailable = true) + + assertIs(plan) + assertEquals("/models/x.gguf", plan.path) + } + + @Test + fun assetWinsOverDownloadWhenNotCached() { + val plan = ModelResolver.plan(targetPath = "/models/x.gguf", targetExists = false, assetAvailable = true) + + assertIs(plan) + assertEquals("/models/x.gguf", plan.path) + } + + @Test + fun downloadsWhenNeitherCachedNorBundled() { + val plan = ModelResolver.plan(targetPath = "/models/x.gguf", targetExists = false, assetAvailable = false) + + assertEquals(ResolutionPlan.Download, plan) + } + + @Test + fun partPathNeverCollidesWithTheFinalTarget() { + assertEquals("/models/x.gguf.part", ModelResolver.partPath("/models/x.gguf")) + } +} diff --git a/KernelRace/shared/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.jvm.kt b/KernelRace/shared/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.jvm.kt new file mode 100644 index 0000000..917f357 --- /dev/null +++ b/KernelRace/shared/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.jvm.kt @@ -0,0 +1,32 @@ +package sk.ainet.samples.kernelrace.engine + +import sk.ainet.apps.llm.OptimizedLLMMode +import sk.ainet.apps.llm.OptimizedLLMRuntime +import sk.ainet.apps.llm.tokenizer.TokenizerFactory +import sk.ainet.context.ExecutionContext +import sk.ainet.io.JvmRandomAccessSource +import sk.ainet.io.model.QuantPolicy +import sk.ainet.lang.types.FP32 +import sk.ainet.models.llama.DecoderGgufWeightLoader +import sk.ainet.models.llama.LlamaNetworkLoader +import sk.ainet.samples.kernelrace.model.ModelData + +/** File-based random-access reads, same shape as the Android fast path but without the + * NEON JNI kernels — desktop has no ARM hardware to dispatch to. */ +actual suspend fun buildLlamaComponents(ctx: ExecutionContext, model: ModelData): LlamaComponents { + val path = (model as ModelData.FilePath).path + val weights = DecoderGgufWeightLoader( + randomAccessProvider = { JvmRandomAccessSource.open(path) }, + quantPolicy = QuantPolicy.NATIVE_OPTIMIZED, + acceptedArchitectures = setOf("llama", "mistral"), + ).loadToMapStreaming(ctx) + val runtime = OptimizedLLMRuntime( + model = LlamaNetworkLoader.fromWeights(weights), + ctx = ctx, + mode = OptimizedLLMMode.DIRECT, + dtype = FP32::class, + bos = weights.metadata.bosTokenId, + ) + val tokenizer = JvmRandomAccessSource.open(path).use { TokenizerFactory.fromGgufSource(it) } + return LlamaComponents(runtime, tokenizer) +} diff --git a/KernelRace/shared/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/model/DesktopModelProvider.kt b/KernelRace/shared/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/model/DesktopModelProvider.kt new file mode 100644 index 0000000..40b4086 --- /dev/null +++ b/KernelRace/shared/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/model/DesktopModelProvider.kt @@ -0,0 +1,81 @@ +package sk.ainet.samples.kernelrace.model + +import java.io.File +import kotlin.time.TimeSource +import kotlinx.io.Buffer +import kotlinx.io.buffered +import kotlinx.io.files.SystemFileSystem +import sk.ainet.data.source.KtorRemoteDataSourceFetcher +import sk.ainet.samples.kernelrace.platform.logEvent +import kotlinx.io.files.Path as KotlinxPath + +/** + * Desktop has no bundled-asset concept, so this is cache-hit-or-download: the model lands in + * `~/.skainet-examples/kernelrace/models/` and is reused across runs. + */ +class DesktopModelProvider( + private val cacheDir: String = defaultCacheDir(), +) : ModelProvider { + + override suspend fun resolve(onProgress: (String) -> Unit): ModelData { + val dir = File(cacheDir).apply { mkdirs() } + val target = File(dir, HF_FILE) + + return when (val plan = ModelResolver.plan(target.path, target.exists(), assetAvailable = false)) { + is ResolutionPlan.UseCached -> { + logEvent("model_cached", "file" to target.name, "fileMB" to target.length() / 1_000_000) + ModelData.FilePath(plan.path) + } + is ResolutionPlan.UseAsset -> error("desktop has no bundled asset") + ResolutionPlan.Download -> { + download(target, onProgress) + ModelData.FilePath(target.path) + } + } + } + + private suspend fun download(target: File, onProgress: (String) -> Unit) { + val url = "https://huggingface.co/$HF_REPO/resolve/main/$HF_FILE" + val tmp = File(target.parentFile, ModelResolver.partPath(target.name)) + val fetcher = KtorRemoteDataSourceFetcher() + logEvent("download_start", "url" to url) + val startedAt = TimeSource.Monotonic.markNow() + var received = 0L + try { + val content = fetcher.fetch(url, emptyMap()) + val totalMb = content.sizeBytes?.let { (it / 1_000_000).toString() } ?: "?" + content.source.use { source -> + SystemFileSystem.sink(KotlinxPath(tmp.path)).buffered().use { sink -> + val chunk = Buffer() + var lastReported = -1L + while (true) { + val n = source.readAtMostTo(chunk, 1024 * 1024) + if (n == -1L) break + sink.write(chunk, n) + received += n + val mb = received / 1_000_000 + if (mb != lastReported) { + lastReported = mb + onProgress("Downloading model… $mb / $totalMb MB") + } + } + } + } + check(tmp.renameTo(target)) { "rename failed: $tmp -> $target" } + logEvent( + "download_done", + "fileMB" to received / 1_000_000, + "durMs" to startedAt.elapsedNow().inWholeMilliseconds, + ) + } catch (e: Exception) { + tmp.delete() + logEvent("download_failed", "receivedMB" to received / 1_000_000, "error" to (e.message ?: e::class.simpleName)) + throw e + } finally { + fetcher.close() + } + } +} + +private fun defaultCacheDir(): String = + File(System.getProperty("user.home"), ".skainet-examples/kernelrace/models").path diff --git a/KernelRace/shared/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.jvm.kt b/KernelRace/shared/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.jvm.kt new file mode 100644 index 0000000..efdde9c --- /dev/null +++ b/KernelRace/shared/src/jvmMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.jvm.kt @@ -0,0 +1,9 @@ +package sk.ainet.samples.kernelrace.platform + +actual fun kernelTierLabel(): String = "JVM (scalar)" + +actual val supportsKernelRace: Boolean = false + +internal actual fun platformLog(line: String) { + println("[SKAINET_PERF_JVM] $line") +} diff --git a/KernelRace/shared/src/jvmTest/kotlin/sk/ainet/samples/kernelrace/engine/GgufSmokeTest.kt b/KernelRace/shared/src/jvmTest/kotlin/sk/ainet/samples/kernelrace/engine/GgufSmokeTest.kt new file mode 100644 index 0000000..65637ae --- /dev/null +++ b/KernelRace/shared/src/jvmTest/kotlin/sk/ainet/samples/kernelrace/engine/GgufSmokeTest.kt @@ -0,0 +1,39 @@ +package sk.ainet.samples.kernelrace.engine + +import java.io.File +import kotlin.test.Test +import kotlinx.coroutines.test.runTest +import sk.ainet.context.DirectCpuExecutionContext +import sk.ainet.samples.kernelrace.model.HF_FILE +import sk.ainet.samples.kernelrace.model.ModelData + +/** + * Loads the real SmolLM2 GGUF and runs a couple of forward passes end to end. Not a hermetic + * unit test — it's gated on the model already sitting in the desktop cache dir (the same place + * [sk.ainet.samples.kernelrace.model.DesktopModelProvider] downloads it to), so a clean CI + * checkout with no cached model skips it rather than failing the build. + */ +class GgufSmokeTest { + + private val cachedModel = File( + System.getProperty("user.home"), + ".skainet-examples/kernelrace/models/$HF_FILE", + ) + + @Test + fun tokenizesAndGeneratesAFewTokens() = runTest { + if (!cachedModel.exists()) { + println("Skipping GgufSmokeTest: no cached model at $cachedModel") + return@runTest + } + + val ctx = DirectCpuExecutionContext() + val engine = LlmEngine.load(ctx, ModelData.FilePath(cachedModel.path)) + + val tokens = mutableListOf() + val text = engine.generate("Say hi in three words.", maxTokens = 8) { tokens.add(it) } + + check(tokens.isNotEmpty()) { "expected at least one streamed token" } + check(text.isNotBlank()) { "expected non-blank generated text" } + } +} diff --git a/KernelRace/shared/src/jvmTest/kotlin/sk/ainet/samples/kernelrace/vm/ChatViewModelTest.kt b/KernelRace/shared/src/jvmTest/kotlin/sk/ainet/samples/kernelrace/vm/ChatViewModelTest.kt new file mode 100644 index 0000000..4cf7993 --- /dev/null +++ b/KernelRace/shared/src/jvmTest/kotlin/sk/ainet/samples/kernelrace/vm/ChatViewModelTest.kt @@ -0,0 +1,140 @@ +package sk.ainet.samples.kernelrace.vm + +import kotlin.test.AfterTest +import kotlin.test.BeforeTest +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertFalse +import kotlin.test.assertTrue +import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.ExperimentalCoroutinesApi +import kotlinx.coroutines.awaitCancellation +import kotlinx.coroutines.test.UnconfinedTestDispatcher +import kotlinx.coroutines.test.resetMain +import kotlinx.coroutines.test.runTest +import kotlinx.coroutines.test.setMain +import sk.ainet.samples.kernelrace.engine.GenerativeEngine + +/** + * [androidx.lifecycle.ViewModel.viewModelScope] launches on `Dispatchers.Main`, and + * `ChatViewModel.generate` hops to its injectable `generationDispatcher` for the actual + * generation call. Both need to run on the SAME virtual-time scheduler as the test body for + * `runTest` to see the launched work complete — an [UnconfinedTestDispatcher] tied to + * `runTest`'s own `testScheduler` does that: everything (Main + generation) executes eagerly, + * on the one scheduler `runTest` already knows how to wait for. + */ +@OptIn(ExperimentalCoroutinesApi::class) +class ChatViewModelTest { + + @BeforeTest + fun setUp() = Dispatchers.setMain(UnconfinedTestDispatcher()) + + @AfterTest + fun tearDown() = Dispatchers.resetMain() + + private class FakeEngine(private val piecesToEmit: List) : GenerativeEngine { + var generateCalls = 0 + override suspend fun generate(prompt: String, maxTokens: Int, onToken: (String) -> Unit): String { + generateCalls++ + piecesToEmit.forEach(onToken) + return piecesToEmit.joinToString("") + } + } + + private fun viewModel( + loadModel: suspend (onProgress: (String) -> Unit) -> GenerativeEngine, + dispatcher: kotlinx.coroutines.CoroutineDispatcher, + ) = ChatViewModel(loadModel = loadModel, generationDispatcher = dispatcher) + + @Test + fun generateStreamsOutputAndEndsIdle() = runTest { + val dispatcher = UnconfinedTestDispatcher(testScheduler) + Dispatchers.setMain(dispatcher) + val engine = FakeEngine(listOf("Hel", "lo")) + val vm = viewModel(loadModel = { engine }, dispatcher = dispatcher) + + vm.generate("hi") + + val state = vm.state.value + assertEquals("Hello", state.output) + assertFalse(state.busy) + assertTrue(state.status.startsWith("Done")) + assertEquals(1, engine.generateCalls) + } + + @Test + fun blankPromptIsIgnored() = runTest { + val dispatcher = UnconfinedTestDispatcher(testScheduler) + Dispatchers.setMain(dispatcher) + val engine = FakeEngine(listOf("x")) + val vm = viewModel(loadModel = { engine }, dispatcher = dispatcher) + + vm.generate(" ") + + assertEquals(0, engine.generateCalls) + assertFalse(vm.state.value.busy) + } + + @Test + fun secondCallWhileBusyIsIgnored() = runTest { + val dispatcher = UnconfinedTestDispatcher(testScheduler) + Dispatchers.setMain(dispatcher) + // A load that never returns keeps the ViewModel busy for the duration of this test. + val engine = FakeEngine(listOf("a")) + val vm = viewModel(loadModel = { awaitCancellation() }, dispatcher = dispatcher) + + vm.generate("first") + vm.generate("second") // dropped: still busy, load hasn't resolved + + assertEquals(0, engine.generateCalls) + assertTrue(vm.state.value.busy) + } + + @Test + fun loadFailureSurfacesAsAnErrorStatus() = runTest { + val dispatcher = UnconfinedTestDispatcher(testScheduler) + Dispatchers.setMain(dispatcher) + val vm = viewModel(loadModel = { error("boom") }, dispatcher = dispatcher) + + vm.generate("hi") + + val state = vm.state.value + assertFalse(state.busy) + assertTrue(state.status.contains("Error")) + } + + @Test + fun engineIsCachedAcrossGenerations() = runTest { + val dispatcher = UnconfinedTestDispatcher(testScheduler) + Dispatchers.setMain(dispatcher) + val engine = FakeEngine(listOf("x")) + var loadCalls = 0 + val vm = viewModel(loadModel = { loadCalls++; engine }, dispatcher = dispatcher) + + vm.generate("first") + vm.generate("second") + + assertEquals(1, loadCalls) + assertEquals(2, engine.generateCalls) + } + + @Test + fun kernelModeChangeDropsTheCachedEngineAndClearsOutput() = runTest { + val dispatcher = UnconfinedTestDispatcher(testScheduler) + Dispatchers.setMain(dispatcher) + val engine = FakeEngine(listOf("x")) + var loadCalls = 0 + val vm = viewModel(loadModel = { loadCalls++; engine }, dispatcher = dispatcher) + + vm.generate("first") + + vm.onKernelModeChanged(newKernelTier = "SCALAR", scalarMode = true) + assertEquals("", vm.state.value.output) + assertEquals("SCALAR", vm.state.value.kernelTier) + assertTrue(vm.state.value.scalarMode) + + vm.generate("second") + + assertEquals(2, loadCalls) // engine reloaded after the mode change + } +} diff --git a/KernelRace/shared/src/wasmJsMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.wasmJs.kt b/KernelRace/shared/src/wasmJsMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.wasmJs.kt new file mode 100644 index 0000000..b688e98 --- /dev/null +++ b/KernelRace/shared/src/wasmJsMain/kotlin/sk/ainet/samples/kernelrace/engine/LlamaRuntimeBuilder.wasmJs.kt @@ -0,0 +1,9 @@ +package sk.ainet.samples.kernelrace.engine + +import sk.ainet.context.ExecutionContext +import sk.ainet.samples.kernelrace.model.ModelData + +/** No filesystem in the browser — the bytes were bundled at build time and read by composeApp + * via `Res.readBytes(...)`, handed down as [ModelData.Bytes]. */ +actual suspend fun buildLlamaComponents(ctx: ExecutionContext, model: ModelData): LlamaComponents = + buildLlamaComponentsFallback(ctx, (model as ModelData.Bytes).bytes) diff --git a/KernelRace/shared/src/wasmJsMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.wasmJs.kt b/KernelRace/shared/src/wasmJsMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.wasmJs.kt new file mode 100644 index 0000000..023dade --- /dev/null +++ b/KernelRace/shared/src/wasmJsMain/kotlin/sk/ainet/samples/kernelrace/platform/Platform.wasmJs.kt @@ -0,0 +1,9 @@ +package sk.ainet.samples.kernelrace.platform + +actual fun kernelTierLabel(): String = "Wasm (scalar)" + +actual val supportsKernelRace: Boolean = false + +internal actual fun platformLog(line: String) { + println("[SKAINET_PERF_WASM] $line") +} diff --git a/KernelRace/webapp.json b/KernelRace/webapp.json new file mode 100644 index 0000000..dd3fc4d --- /dev/null +++ b/KernelRace/webapp.json @@ -0,0 +1,14 @@ +{ + "id": "kernelrace", + "name": "Kernel Race", + "description": "On-device LLM chat with a NEON-vs-scalar kernel race on Android; this web build runs SKaiNET's scalar fallback path.", + "platforms": ["android", "desktop", "wasm"], + "distDirs": [ + "composeApp/build/dist/wasmJs/productionExecutable" + ], + "build": { + "command": "./gradlew :composeApp:wasmJsBrowserDistribution", + "description": "Builds the wasm browser distribution, including the bundled ~145 MB GGUF model" + }, + "screenshot": "docs/screenshots/split_race.png" +} diff --git a/KllamaDemo/gradle/libs.versions.toml b/KllamaDemo/gradle/libs.versions.toml index c3cf62f..aafed56 100644 --- a/KllamaDemo/gradle/libs.versions.toml +++ b/KllamaDemo/gradle/libs.versions.toml @@ -12,7 +12,7 @@ androidx-testExt = "1.3.0" composeHotReload = "1.0.0" composeMultiplatform = "1.10.1" junit = "4.13.2" -kotlin = "2.3.21" +kotlin = "2.4.10" kotlinx-browser = "0.5.0" kotlinx-coroutines = "1.11.0" kotlinx-datetime = "0.7.1" @@ -20,8 +20,8 @@ kotlinxIo = "0.9.0" ktor = "3.4.0" logback = "1.5.29" material3 = "1.10.0-alpha05" -skainet = "0.34.0" -skainet-transformers = "0.34.1" +skainet = "0.39.1" +skainet-transformers = "0.39.1" [libraries] kotlin-test = { module = "org.jetbrains.kotlin:kotlin-test", version.ref = "kotlin" } diff --git a/MNISTDemo/gradle/libs.versions.toml b/MNISTDemo/gradle/libs.versions.toml index 6ffb75a..4b27c24 100644 --- a/MNISTDemo/gradle/libs.versions.toml +++ b/MNISTDemo/gradle/libs.versions.toml @@ -15,11 +15,11 @@ androidx-testExt = "1.2.1" composeHotReload = "1.0.0" composeMultiplatform = "1.10.1" junit = "4.13.2" -kotlin = "2.3.21" +kotlin = "2.4.10" kotlinx-coroutines = "1.10.2" ktor = "3.1.3" logback = "1.5.18" -skainet = "0.34.0" +skainet = "0.40.1" kotlinxIo = "0.8.2" ktorClientCore = "3.3.3" ktorClientPlugins = "3.1.1" @@ -44,7 +44,7 @@ skainet-compile-dag = { module = "sk.ainet.core:skainet-compile-dag", version.re skainet-backend-cpu = { module = "sk.ainet.core:skainet-backend-cpu", version.ref = "skainet" } skainet-backend-cpu-jvm = { module = "sk.ainet.core:skainet-backend-cpu", version.ref = "skainet" } skainet-data-api = { module = "sk.ainet.core:skainet-data-api", version.ref = "skainet" } -skainet-data-simple = { module = "sk.ainet.core:skainet-data-basic", version.ref = "skainet" } +skainet-data-simple = { module = "sk.ainet.core:skainet-data-simple", version.ref = "skainet" } skainet-io-core = { module = "sk.ainet.core:skainet-io-core", version.ref = "skainet" } skainet-io-gguf = { module = "sk.ainet.core:skainet-io-gguf", version.ref = "skainet" } skainet-io-onnx = { module = "sk.ainet.core:skainet-io-onnx", version.ref = "skainet" } diff --git a/MNISTDemo/webapp.json b/MNISTDemo/webapp.json index 01fc492..328f69c 100644 --- a/MNISTDemo/webapp.json +++ b/MNISTDemo/webapp.json @@ -2,6 +2,7 @@ "id": "mnistdemo", "name": "MNIST Demo", "description": "Handwritten digit classifier with in-app training using SKaiNET ML framework", + "platforms": ["android", "ios", "desktop", "wasm"], "distDirs": [ "composeApp/build/kotlin-webpack/wasmJs/productionExecutable", "composeApp/build/processedResources/wasmJs/main" diff --git a/MnistJavaDemo/build.gradle.kts b/MnistJavaDemo/build.gradle.kts index 992543f..fcdbd8b 100644 --- a/MnistJavaDemo/build.gradle.kts +++ b/MnistJavaDemo/build.gradle.kts @@ -1,7 +1,7 @@ plugins { java application - kotlin("jvm") version "2.3.21" + kotlin("jvm") version "2.4.10" } java { diff --git a/MnistJavaDemo/gradle/libs.versions.toml b/MnistJavaDemo/gradle/libs.versions.toml index 8f41104..d593767 100644 --- a/MnistJavaDemo/gradle/libs.versions.toml +++ b/MnistJavaDemo/gradle/libs.versions.toml @@ -1,6 +1,6 @@ [versions] -skainet = "0.34.0" -kotlin = "2.3.21" +skainet = "0.40.1" +kotlin = "2.4.10" kotlinx-coroutines = "1.10.2" kotlinx-io = "0.8.2" junit = "4.13.2" @@ -11,7 +11,7 @@ skainet-lang-core = { module = "sk.ainet.core:skainet-lang-core-jvm", version.re skainet-lang-models = { module = "sk.ainet.core:skainet-lang-models-jvm", version.ref = "skainet" } skainet-backend-cpu = { module = "sk.ainet.core:skainet-backend-cpu-jvm", version.ref = "skainet" } skainet-data-api = { module = "sk.ainet.core:skainet-data-api-jvm", version.ref = "skainet" } -skainet-data-simple = { module = "sk.ainet.core:skainet-data-basic-jvm", version.ref = "skainet" } +skainet-data-simple = { module = "sk.ainet.core:skainet-data-simple-jvm", version.ref = "skainet" } skainet-data-transform = { module = "sk.ainet.core:skainet-data-transform-jvm", version.ref = "skainet" } skainet-io-core = { module = "sk.ainet.core:skainet-io-core-jvm", version.ref = "skainet" } skainet-io-image = { module = "sk.ainet.core:skainet-io-image-jvm", version.ref = "skainet" } diff --git a/MnistJavaDemo/webapp.json b/MnistJavaDemo/webapp.json new file mode 100644 index 0000000..5ad4dff --- /dev/null +++ b/MnistJavaDemo/webapp.json @@ -0,0 +1,6 @@ +{ + "id": "mnistjavademo", + "name": "MNIST Java Demo", + "description": "The same MNIST digit detector as a pure-Java CLI — proof of SKaiNET's first-class Java interop from a Kotlin Multiplatform engine.", + "platforms": ["cli", "java"] +} diff --git a/README.md b/README.md index eec5254..792bcd8 100644 --- a/README.md +++ b/README.md @@ -67,6 +67,19 @@ language toggle. --- +### 🏁 Kernel Race + +On-device LLM chat accelerated by SKaiNET's hand-written ARM NEON kernels on Android, with a +built-in **NEON vs scalar** A/B comparison — including a one-phone split-screen race. The same +Kotlin codebase also runs on Desktop and in the browser, showing the other kernel tiers SKaiNET +falls back to without NEON hardware. + +Kernel Race — split-screen NEON vs scalar generation + +📂 [`KernelRace/`](KernelRace/) · runs on Android · Desktop · Wasm + +--- + ### 💬 Kllama Demo A Qwen3 LLM **playground in the browser** — chat, completion, translation, tool diff --git a/SinusApproximator/CHANGELOG.md b/SinusApproximator/CHANGELOG.md index 10db63c..24731a2 100644 --- a/SinusApproximator/CHANGELOG.md +++ b/SinusApproximator/CHANGELOG.md @@ -5,6 +5,12 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [0.40.1] - 2026-08-12 + +### Changed +- Update to SKaiNET 0.40.1 from Maven Central. +- Bump Kotlin to 2.4.10 to match the compiler SKaiNET's 0.40.1 klibs are built with. + ## [0.34.0] - 2026-07-07 ### Changed diff --git a/SinusApproximator/gradle/libs.versions.toml b/SinusApproximator/gradle/libs.versions.toml index a50aafc..6e5a650 100644 --- a/SinusApproximator/gradle/libs.versions.toml +++ b/SinusApproximator/gradle/libs.versions.toml @@ -19,11 +19,11 @@ composeHotReload = "1.0.0" # (getAdapterMaxTextureSize) at first render on macOS/Metal. composeMultiplatform = "1.10.1" junit = "4.13.2" -kotlin = "2.3.21" +kotlin = "2.4.10" kotlinx-coroutines = "1.10.2" ktor = "3.1.3" logback = "1.5.18" -skainet = "0.34.0" +skainet = "0.40.1" kotlinxIo = "0.8.2" [libraries] @@ -45,7 +45,7 @@ skainet-compile-dag = { module = "sk.ainet.core:skainet-compile-dag", version.re skainet-backend-cpu = { module = "sk.ainet.core:skainet-backend-cpu", version.ref = "skainet" } skainet-backend-cpu-jvm = { module = "sk.ainet.core:skainet-backend-cpu", version.ref = "skainet" } skainet-data-api = { module = "sk.ainet.core:skainet-data-api", version.ref = "skainet" } -skainet-data-simple = { module = "sk.ainet.core:skainet-data-basic", version.ref = "skainet" } +skainet-data-simple = { module = "sk.ainet.core:skainet-data-simple", version.ref = "skainet" } skainet-io-core = { module = "sk.ainet.core:skainet-io-core", version.ref = "skainet" } skainet-io-gguf = { module = "sk.ainet.core:skainet-io-gguf", version.ref = "skainet" } skainet-io-onnx = { module = "sk.ainet.core:skainet-io-onnx", version.ref = "skainet" } diff --git a/SinusApproximator/webapp.json b/SinusApproximator/webapp.json index 41cf038..fe67dbc 100644 --- a/SinusApproximator/webapp.json +++ b/SinusApproximator/webapp.json @@ -2,6 +2,7 @@ "id": "sinusapproximator", "name": "Sinus Approximator", "description": "Sinus Approximator as web app", + "platforms": ["android", "ios", "desktop", "wasm"], "distDirs": [ "composeApp/build/kotlin-webpack/wasmJs/productionExecutable", "composeApp/build/processedResources/wasmJs/main" diff --git a/TinyTransformer/gradle/libs.versions.toml b/TinyTransformer/gradle/libs.versions.toml index e5e50ef..f14ae6f 100644 --- a/TinyTransformer/gradle/libs.versions.toml +++ b/TinyTransformer/gradle/libs.versions.toml @@ -12,13 +12,13 @@ composeHotReload = "1.0.0" # classpath, which crashes with UnsatisfiedLinkError # (getAdapterMaxTextureSize) at first render on macOS/Metal. composeMultiplatform = "1.10.1" -# Must match the Kotlin version SKaiNET was built with: SKaiNET 0.34.0 ships -# klibs compiled with Kotlin 2.3.21, and an older compiler cannot read them. -kotlin = "2.3.21" +# Must match the Kotlin version SKaiNET was built with: SKaiNET 0.40.1 ships +# klibs compiled with Kotlin 2.4.10, and an older compiler cannot read them. +kotlin = "2.4.10" kotlinx-coroutines = "1.10.2" # SKaiNET BOM version — published to Maven Central. The BOM pins every # sk.ainet.core:* artifact so individual libraries are declared without versions. -skainet = "0.34.0" +skainet = "0.40.1" [libraries] kotlin-test = { module = "org.jetbrains.kotlin:kotlin-test", version.ref = "kotlin" } diff --git a/TinyTransformer/webapp.json b/TinyTransformer/webapp.json index 8d8cd63..fbd4ef9 100644 --- a/TinyTransformer/webapp.json +++ b/TinyTransformer/webapp.json @@ -2,6 +2,7 @@ "id": "tinytransformer", "name": "Tiny Transformer (KI-ENNA)", "description": "Train a tiny decoder-only transformer live in the browser — inspired by the KI-ENNA project (statistical-thinking.de)", + "platforms": ["android", "ios", "desktop", "wasm"], "distDirs": [ "composeApp/build/kotlin-webpack/wasmJs/productionExecutable", "composeApp/build/processedResources/wasmJs/main" diff --git a/scripts/generate-samples-page.py b/scripts/generate-samples-page.py index 8d58f88..ffa0003 100755 --- a/scripts/generate-samples-page.py +++ b/scripts/generate-samples-page.py @@ -133,6 +133,7 @@ def find_webapp_configs(root_dir: Path, repo_url: str | None = None, branch: str "description": meta.get("description", ""), "screenshot": meta.get("screenshot"), "distDirs": meta.get("distDirs", []), + "platforms": meta.get("platforms", ["web"]), "sourceUrl": source_url, "project_root": project_root }) @@ -461,7 +462,9 @@ def generate_html(apps: list, release_tag: str = "", base_url: str = "https://ex "description": app["description"], "screenshot": app.get("screenshot_url") or app.get("screenshot"), "sourceUrl": app.get("sourceUrl"), - "demoUrl": f"./{app['id']}/" + "platforms": app.get("platforms", ["web"]), + # Only samples that actually ship a web dist get a live demo link. + "demoUrl": f"./{app['id']}/" if app.get("distDirs") else None } for app in apps ], indent=2) @@ -886,6 +889,27 @@ def generate_html(apps: list, release_tag: str = "", base_url: str = "https://ex line-height: 1.5; }} + .platform-chips {{ + display: flex; + flex-wrap: wrap; + gap: 0.375rem; + margin-top: 0.75rem; + }} + + .platform-chip {{ + display: inline-flex; + align-items: center; + padding: 0.2rem 0.55rem; + font-size: 0.625rem; + font-weight: 500; + text-transform: capitalize; + color: hsl(var(--muted-foreground)); + background-color: hsl(var(--muted) / 0.6); + border: 1px solid hsl(var(--border)); + border-radius: 999px; + white-space: nowrap; + }} + /* Card Footer - Stack on mobile, side-by-side on larger */ .card-footer {{ position: relative; @@ -1078,10 +1102,17 @@ def generate_html(apps: list, release_tag: str = "", base_url: str = "https://ex ${{playIcon}} Try Demo ` - : ``; + : (project.sourceUrl + ? ` + How to run + ` + : ``); + + const platformChips = (project.platforms || ['web']).map(p => + `${{escapeHtml(p)}}` + ).join(''); return `
@@ -1095,6 +1126,7 @@ def generate_html(apps: list, release_tag: str = "", base_url: str = "https://ex ${{escapeHtml(project.id)}}

${{escapeHtml(project.description)}}

+
${{platformChips}}