- ModelConfig.filename 기본값 gemma4_e2b_q4.bin → gemma-4-E2B-it.litertlm (#342 근본 원인) - _migrateLegacyPath: 기존 .bin 설치본 rename + meta 갱신 (재다운로드 2.4GB 방지, 멱등) - maxTokens 2048 유지 + 1024 회귀 방지 주석 (KV cache < prefill signature 시 추론 실패) - LLM_BACKEND dart-define: 에뮬레이터 CPU 강제 (SwiftShader GPU 네이티브 SIGABRT 회피) - GEMMA_MODEL_URL dart-define: 호스트 로컬 모델 서버 주입 (기본값 = HF URL 불변) - debug 전용 cleartext manifest (10.0.2.2 모델 서버용, release 불변) - 마이그레이션 테스트 4건 신규 (AC-2/3/4). 171 passed, analyze clean 2026-07-14 에뮬레이터 E2E 검증: 다운로드→로드→tool call 왕복 전 구간 성공. Refs #657, #342 Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
293 lines
9.3 KiB
Dart
293 lines
9.3 KiB
Dart
import 'dart:io';
|
|
|
|
import 'package:flutter/foundation.dart';
|
|
import 'package:flutter_gemma/flutter_gemma.dart';
|
|
|
|
import '../../ai/tools/tool_definition.dart' as tools;
|
|
import 'llm_service.dart';
|
|
|
|
/// HuggingFace access token injected at build time via
|
|
/// `--dart-define=HF_TOKEN=hf_xxx`. Empty string is permitted —
|
|
/// flutter_gemma will only need it for the initial network download,
|
|
/// which our `ModelLifecycle` handles separately; activation from a
|
|
/// local file path generally does not require the token.
|
|
const String _hfToken = String.fromEnvironment('HF_TOKEN', defaultValue: '');
|
|
|
|
/// Backend override for dev builds. The Android emulator advertises a
|
|
/// software Vulkan adapter (SwiftShader) that SIGABRTs LiteRT-LM's GPU
|
|
/// init natively — the Dart-level gpu→cpu fallback can't catch a native
|
|
/// abort, so emulator runs must force CPU: `--dart-define=LLM_BACKEND=cpu`.
|
|
/// Unset (default) keeps the SDK's gpu→cpu fallback for real devices.
|
|
const String _backendOverride =
|
|
String.fromEnvironment('LLM_BACKEND', defaultValue: '');
|
|
|
|
PreferredBackend? get _preferredBackend => switch (_backendOverride) {
|
|
'cpu' => PreferredBackend.cpu,
|
|
'gpu' => PreferredBackend.gpu,
|
|
_ => null,
|
|
};
|
|
|
|
/// One-shot guard so [FlutterGemma.initialize] runs at most once per
|
|
/// isolate. Re-init is unsupported by the underlying plugin.
|
|
bool _initialized = false;
|
|
|
|
/// Real on-device LLM backend using flutter_gemma 0.16.5 + Gemma 4 E2B.
|
|
///
|
|
/// Wired into the existing #215 pipeline: `ModelLifecycle` downloads &
|
|
/// SHA-verifies the .litertlm file, then [load] registers that file with
|
|
/// flutter_gemma as the active model. [generateStructured] opens a
|
|
/// short-lived chat with a single [Tool] (Gemma 4 native function
|
|
/// calling) and returns the first matching [FunctionCallResponse]'s args.
|
|
///
|
|
/// Function-calling design notes (see fn-gemma_llm_service.md §B v2):
|
|
/// - Gemma 4 SDK injects the tool declaration via its chat template, so
|
|
/// we pass [Tool] to `createChat(tools: ...)` rather than appending a
|
|
/// schema instruction to the prompt (double-wrap risk).
|
|
/// - `ToolChoice.required` forces the model to emit a function call.
|
|
class GemmaLlmService implements LlmService {
|
|
final String modelPath;
|
|
|
|
GemmaLlmService({required this.modelPath});
|
|
|
|
InferenceModel? _model;
|
|
bool _loaded = false;
|
|
Future<void>? _loadingFuture;
|
|
|
|
@override
|
|
bool get isLoaded => _loaded;
|
|
|
|
/// #311 AC7: concurrent-call guard. If a load is already in-flight (e.g.
|
|
/// `ChatScreen` warm-up + a racing `userTurn` lazy load), return the same
|
|
/// Future so native init runs at most once per process.
|
|
/// See `docs/design/311-llm-warmup/fn-concurrent_load_guard.md`.
|
|
@override
|
|
Future<void> load() {
|
|
if (_loaded) return Future.value();
|
|
final existing = _loadingFuture;
|
|
if (existing != null) return existing;
|
|
final future = _doLoad();
|
|
_loadingFuture = future;
|
|
return future.whenComplete(() {
|
|
_loadingFuture = null;
|
|
});
|
|
}
|
|
|
|
Future<void> _doLoad() async {
|
|
if (!await File(modelPath).exists()) {
|
|
throw FileSystemException('model file missing', modelPath);
|
|
}
|
|
if (!_initialized) {
|
|
await FlutterGemma.initialize(huggingFaceToken: _hfToken);
|
|
_initialized = true;
|
|
}
|
|
await FlutterGemma.installModel(
|
|
modelType: ModelType.gemma4,
|
|
fileType: ModelFileType.litertlm,
|
|
).fromFile(modelPath).install();
|
|
// #342 root cause was the model *filename* (.bin — LiteRT-LM rejects it;
|
|
// must be .litertlm), NOT the KV cache size. maxTokens stays 2048: the
|
|
// Gemma 4 E2B compiled graph requires a cache ≥ its prefill signature —
|
|
// 1024 fails tensor allocation (DYNAMIC_UPDATE_SLICE prepare, verified
|
|
// on-emulator 2026-07-09).
|
|
final model = await FlutterGemma.getActiveModel(
|
|
maxTokens: 2048,
|
|
preferredBackend: _preferredBackend,
|
|
);
|
|
_model = model;
|
|
_loaded = true;
|
|
}
|
|
|
|
@override
|
|
Future<void> unload() async {
|
|
final m = _model;
|
|
_model = null;
|
|
_loaded = false;
|
|
if (m != null) {
|
|
try {
|
|
await m.close();
|
|
} catch (_) {
|
|
// Best-effort cleanup — runtime may already be torn down.
|
|
}
|
|
}
|
|
}
|
|
|
|
@override
|
|
Future<Map<String, dynamic>> generateStructured(
|
|
String prompt,
|
|
Map<String, dynamic> schema,
|
|
) async {
|
|
if (!_loaded || _model == null) {
|
|
throw StateError('LlmService not loaded');
|
|
}
|
|
final fnName = schema['name'];
|
|
final fnParams = schema['parameters'];
|
|
if (fnName is! String || fnName.isEmpty) {
|
|
throw ArgumentError('schema.name missing');
|
|
}
|
|
if (fnParams is! Map) {
|
|
throw ArgumentError('schema.parameters missing');
|
|
}
|
|
final fnDesc = (schema['description'] as String?) ?? '';
|
|
final tool = Tool(
|
|
name: fnName,
|
|
description: fnDesc,
|
|
parameters: Map<String, dynamic>.from(fnParams),
|
|
);
|
|
|
|
final chat = await _model!.createChat(
|
|
modelType: ModelType.gemma4,
|
|
supportsFunctionCalls: true,
|
|
toolChoice: ToolChoice.required,
|
|
tools: [tool],
|
|
);
|
|
try {
|
|
await chat.addQueryChunk(Message.text(text: prompt, isUser: true));
|
|
final stream = chat.generateChatResponseAsync();
|
|
return await collectFunctionCall(stream, fnName);
|
|
} finally {
|
|
try {
|
|
await chat.close();
|
|
} catch (_) {
|
|
// Native session close failure is non-fatal — log + continue.
|
|
}
|
|
}
|
|
}
|
|
|
|
@override
|
|
Future<LlmChatSession> startChat({
|
|
required List<tools.ToolDefinition> tools,
|
|
}) async {
|
|
if (!_loaded || _model == null) {
|
|
throw StateError('LlmService not loaded');
|
|
}
|
|
final gemmaTools = tools
|
|
.map((t) => Tool(
|
|
name: t.name,
|
|
description: t.description,
|
|
parameters: Map<String, dynamic>.from(t.parametersSchema),
|
|
))
|
|
.toList();
|
|
final chat = await _model!.createChat(
|
|
modelType: ModelType.gemma4,
|
|
supportsFunctionCalls: true,
|
|
// ToolChoice.auto = 모델이 자율 결정 (multi-tool + reply-only 모두 지원).
|
|
toolChoice: ToolChoice.auto,
|
|
tools: gemmaTools,
|
|
);
|
|
return _GemmaChatSession(chat);
|
|
}
|
|
}
|
|
|
|
class _GemmaChatSession implements LlmChatSession {
|
|
final dynamic _chat;
|
|
bool _closed = false;
|
|
|
|
_GemmaChatSession(this._chat);
|
|
|
|
@override
|
|
Stream<LlmChatEvent> sendUser(String text) {
|
|
if (_closed) {
|
|
throw StateError('LlmChatSession is closed');
|
|
}
|
|
return _run(Message.text(text: text, isUser: true));
|
|
}
|
|
|
|
@override
|
|
Stream<LlmChatEvent> sendToolResult({
|
|
required String toolName,
|
|
required Map<String, dynamic> result,
|
|
}) {
|
|
if (_closed) {
|
|
throw StateError('LlmChatSession is closed');
|
|
}
|
|
return _run(Message.toolResponse(toolName: toolName, response: result));
|
|
}
|
|
|
|
Stream<LlmChatEvent> _run(Message msg) async* {
|
|
await _chat.addQueryChunk(msg);
|
|
final Stream<ModelResponse> stream = _chat.generateChatResponseAsync();
|
|
await for (final event in stream) {
|
|
if (event is TextResponse) {
|
|
yield LlmTextChunk(event.token);
|
|
} else if (event is FunctionCallResponse) {
|
|
yield LlmFunctionCall(
|
|
event.name,
|
|
Map<String, dynamic>.from(event.args),
|
|
);
|
|
return; // model hands control back to caller for tool exec
|
|
} else if (event is ParallelFunctionCallResponse &&
|
|
event.calls.isNotEmpty) {
|
|
// ADR-0005: parallel calls collapsed to first — sequential dispatch.
|
|
final first = event.calls.first;
|
|
yield LlmFunctionCall(
|
|
first.name,
|
|
Map<String, dynamic>.from(first.args),
|
|
);
|
|
return;
|
|
}
|
|
// ThinkingResponse / other: skip.
|
|
}
|
|
}
|
|
|
|
@override
|
|
Future<void> close() async {
|
|
if (_closed) return;
|
|
_closed = true;
|
|
try {
|
|
await _chat.close();
|
|
} catch (_) {
|
|
// Best-effort cleanup.
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Extracts the first `FunctionCallResponse(name == expectedName)` from
|
|
/// a flutter_gemma response stream. `TextResponse` / `ThinkingResponse`
|
|
/// events are skipped. A mismatched name throws fast.
|
|
///
|
|
/// File-private under `_collectFunctionCall` from [GemmaLlmService];
|
|
/// exposed as a top-level via `@visibleForTesting` so unit tests can
|
|
/// feed synthetic streams (see fn-spec §D, 8 test cases).
|
|
@visibleForTesting
|
|
Future<Map<String, dynamic>> collectFunctionCall(
|
|
Stream<ModelResponse> stream,
|
|
String expectedName,
|
|
) async {
|
|
Map<String, dynamic>? result;
|
|
String? wrongName;
|
|
try {
|
|
await for (final event in stream) {
|
|
if (event is FunctionCallResponse) {
|
|
if (event.name == expectedName) {
|
|
result = Map<String, dynamic>.from(event.args);
|
|
break;
|
|
} else {
|
|
wrongName = event.name;
|
|
break;
|
|
}
|
|
}
|
|
if (event is ParallelFunctionCallResponse && event.calls.isNotEmpty) {
|
|
final first = event.calls.first;
|
|
if (first.name == expectedName) {
|
|
result = Map<String, dynamic>.from(first.args);
|
|
} else {
|
|
wrongName = first.name;
|
|
}
|
|
break;
|
|
}
|
|
// TextResponse / ThinkingResponse: skip.
|
|
}
|
|
} catch (_) {
|
|
// Discard raw error to avoid leaking prompt content in logs/crash
|
|
// reports — the caller surfaces a generic message.
|
|
throw const FormatException('stream error');
|
|
}
|
|
if (wrongName != null) {
|
|
throw FormatException('unexpected function: $wrongName');
|
|
}
|
|
if (result == null) {
|
|
throw const FormatException('no function call emitted');
|
|
}
|
|
return result;
|
|
}
|