- v2.2.299 아키텍처 수렴: coreChat 통일(엔진 휴리스틱 3벌 제거), 기업 모드 검색 오케스트레이터 승격, lib/execUtil(실행 래퍼 6곳·Python 탐지 3벌 단일화), lib/kstSchedule(워처 4개 nowInKst 통합), estimateTokens 통합, 설정 접근 규칙 명문화 - v2.2.300 채팅 화면 정리: LiveReasoningFilter(스트리밍 중 <think>/Harmony 추론 토큰 단위 차단), 확신도·검토요청 footer 기본 숨김(계산·Reflection 은 유지) - v2.2.301 문맥·의도 이해: [답변 전 이해 원칙] 상시 주입, 워크플로우 의도 브리핑, Report QA 루프(규칙 레지스트리+실측치+회귀 게이트, 블로그_v3 개념 이식) - v2.2.302 /benchmark 비즈니스 렌즈(가격·수익·운영)+빌드 프롬프트 모드+QA 연계 - v2.2.303 handoff 모드(측정치 무손실 인수인계 문서)+/claude(Claude Code 터미널 위임) - v2.2.304 이식성: 지식 경로 두뇌-상대 규약(pickWikiDir 상대 해석), 이사 체크리스트 - v2.2.305 Claude 구독 엔진: claude: 프로바이더(CLI 위임, 모델 드롭다운 자동 노출, coreChat 지원 — 워크플로우·QA도 구독 모델 가능) - v2.2.306 Tone Guard: AI 상투어 금지 레지스트리(상담사 화법 실사례 8종+대조 예시) - v2.2.307 /benchmark 레이아웃 골격(sectionRoles 결정론 분석, 롤링 배너 즉답), 파트별 실패 격리, 합성 타임아웃 120→300초 검증: tsc 무오류 + jest 888 통과 + esbuild 정상 Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
219 lines
9.5 KiB
TypeScript
219 lines
9.5 KiB
TypeScript
import * as fs from 'fs';
|
|
import * as path from 'path';
|
|
import { getConfig } from '../config';
|
|
import { buildApiUrl, logError, logInfo, resolveEngine, summarizeText, _getBrainDir } from '../utils';
|
|
|
|
/**
|
|
* IAIService: AI 모델 호출에 대한 인터페이스.
|
|
*
|
|
* `call(prompt)` 는 plain user 메시지 1개만 보내는 legacy shortcut이고,
|
|
* `chat({ system, user })` 는 role-aware 호출이다. Telegram 핸들러처럼
|
|
* 모델을 grounding 해야 하는 경로에서는 system을 반드시 채워야 한다 —
|
|
* gemma 같은 작은 모델은 system이 없으면 짧은/모호한 입력에 대해
|
|
* "시는 못 써드려요" 같은 환각 거절을 하는 경향이 있다.
|
|
*/
|
|
export interface IAIService {
|
|
call(prompt: string): Promise<string>;
|
|
chat(req: AIChatRequest): Promise<AIChatResult>;
|
|
}
|
|
|
|
export interface AIChatRequest {
|
|
/** Optional system prompt. Strongly recommended for short / ambiguous user inputs. */
|
|
system?: string;
|
|
/** Required. The user message. */
|
|
user: string;
|
|
/** Optional override (default = config.defaultModel). */
|
|
model?: string;
|
|
/** Optional override (default = config.timeout). */
|
|
timeoutMs?: number;
|
|
/** 샘플링 온도 (기본 0.7). 판정·재순위 등 결정적 작업은 0.0~0.2 권장. */
|
|
temperature?: number;
|
|
/** 출력 토큰 상한 (ollama num_predict / lmstudio max_tokens). 미지정 시 서버 기본. */
|
|
maxTokens?: number;
|
|
/** [ollama 전용] 컨텍스트 창 크기(num_ctx). */
|
|
numCtx?: number;
|
|
/**
|
|
* 외부 abort signal. fetch 가 받는 signal 과 OR 로 결합되어, 사용자가 회사 모드
|
|
* 도중 Stop 을 누르면 진행 중인 generation 이 즉시 중단된다. 없으면 timeout 만
|
|
* 적용. dispatcher 같은 긴 multi-turn 경로에서 반드시 전달할 것.
|
|
*/
|
|
signal?: AbortSignal;
|
|
}
|
|
|
|
export interface AIChatResult {
|
|
content: string;
|
|
/** Engine that actually returned the content. */
|
|
engine: 'lmstudio' | 'ollama' | 'claude-code';
|
|
model: string;
|
|
/** True iff content came back empty after all retries. Caller decides UX. */
|
|
empty: boolean;
|
|
}
|
|
|
|
/**
|
|
* IBrainService: 지식 베이스(Brain) 조작에 대한 인터페이스
|
|
*/
|
|
export interface IBrainService {
|
|
inject(title: string, markdown: string): Promise<string>;
|
|
}
|
|
|
|
/**
|
|
* AIService: Ollama 및 LM Studio 폴백 로직을 포함한 AI 호출 구현체.
|
|
*
|
|
* Behavior:
|
|
* 1. Try the user-configured engine first; on transport / 5xx / empty response,
|
|
* fall through to the other engine.
|
|
* 2. Empty responses are treated as a soft failure: we log + retry the other
|
|
* engine before giving up. Pure exceptions (network blip) trigger the same
|
|
* fallback path.
|
|
* 3. The legacy `call(prompt)` is preserved as a thin wrapper around `chat()`
|
|
* for callers that don't have a system prompt — but new code should pass
|
|
* a system prompt explicitly.
|
|
*/
|
|
export class AIService implements IAIService {
|
|
public async call(prompt: string): Promise<string> {
|
|
const result = await this.chat({ user: prompt });
|
|
return result.content;
|
|
}
|
|
|
|
public async chat(req: AIChatRequest): Promise<AIChatResult> {
|
|
const config = getConfig();
|
|
const model = (req.model || config.defaultModel || '').trim() || 'gemma4:e2b';
|
|
const timeoutMs = req.timeoutMs ?? config.timeout;
|
|
|
|
// [v2.2.305] Claude 구독 모델 ('claude:sonnet' 등) — 로컬 엔진 대신 Claude Code CLI 위임.
|
|
// 워크플로우·Report QA·리랭크 등 coreChat 소비자 전부가 구독 모델로 동작 가능해진다.
|
|
// (다른 클라우드 prefix 는 스트리밍 전용 경로만 지원 — 여기 오면 아래 로컬 시도가 실패로 드러남)
|
|
if (model.startsWith('claude:')) {
|
|
const { runClaudeCode } = await import('../features/providers/claudeCode');
|
|
const messages: Array<{ role: 'system' | 'user' | 'assistant'; content: string }> = [];
|
|
if (req.system && req.system.trim()) messages.push({ role: 'system', content: req.system });
|
|
messages.push({ role: 'user', content: req.user });
|
|
const r = await runClaudeCode({
|
|
messages,
|
|
model: model.slice('claude:'.length),
|
|
signal: req.signal,
|
|
timeoutMs: Math.max(timeoutMs, 120_000),
|
|
});
|
|
return { content: r.text, engine: 'claude-code', model, empty: !r.text.trim() };
|
|
}
|
|
const primaryEngine = resolveEngine(config.ollamaUrl);
|
|
const engines = primaryEngine === 'lmstudio'
|
|
? ['lmstudio', 'ollama'] as const
|
|
: ['ollama', 'lmstudio'] as const;
|
|
|
|
const messages: Array<{ role: 'system' | 'user' | 'assistant'; content: string }> = [];
|
|
if (req.system && req.system.trim()) {
|
|
messages.push({ role: 'system', content: req.system });
|
|
}
|
|
messages.push({ role: 'user', content: req.user });
|
|
|
|
let lastError: Error | null = null;
|
|
let lastEmptyEngine: typeof engines[number] | null = null;
|
|
|
|
for (const engine of engines) {
|
|
const apiUrl = buildApiUrl(config.ollamaUrl, engine, 'chat');
|
|
const temperature = req.temperature ?? 0.7;
|
|
const payload = {
|
|
model,
|
|
messages,
|
|
stream: false,
|
|
...(engine === 'ollama'
|
|
? { options: {
|
|
temperature,
|
|
...(req.maxTokens != null ? { num_predict: req.maxTokens } : {}),
|
|
...(req.numCtx != null ? { num_ctx: req.numCtx } : {}),
|
|
} }
|
|
: {
|
|
temperature,
|
|
...(req.maxTokens != null ? { max_tokens: req.maxTokens } : {}),
|
|
}),
|
|
};
|
|
|
|
try {
|
|
logInfo('[AIService] Request started.', {
|
|
engine, apiUrl, model,
|
|
hasSystem: !!req.system, userChars: req.user.length,
|
|
});
|
|
// timeout signal + 외부 abort signal 결합. 외부 signal 이 fire 되면
|
|
// 진행 중인 fetch 가 즉시 중단되어 사용자 Stop 이 LLM generation 중에도 효과.
|
|
const timeoutSignal = AbortSignal.timeout(timeoutMs);
|
|
const combinedSignal = req.signal
|
|
? AbortSignal.any([req.signal, timeoutSignal])
|
|
: timeoutSignal;
|
|
const res = await fetch(apiUrl, {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify(payload),
|
|
signal: combinedSignal,
|
|
});
|
|
|
|
const rawText = await res.text();
|
|
if (!res.ok) {
|
|
lastError = new Error(`AI call failed: ${res.status} ${summarizeText(rawText, 250)}`);
|
|
logError(`[AIService] ${engine} HTTP ${res.status}`, { body: summarizeText(rawText, 250) });
|
|
continue;
|
|
}
|
|
|
|
const data = rawText ? JSON.parse(rawText) as any : {};
|
|
const content = engine === 'lmstudio'
|
|
? (data.choices?.[0]?.message?.content || '')
|
|
: (data.message?.content || data.response || '');
|
|
|
|
if (!content || !content.trim()) {
|
|
// Treat empty as soft failure so the other engine gets a chance.
|
|
lastEmptyEngine = engine;
|
|
lastError = new Error(`AI engine '${engine}' returned an empty response.`);
|
|
logError(`[AIService] ${engine} empty response — falling through.`, { model });
|
|
continue;
|
|
}
|
|
|
|
return { content, engine, model, empty: false };
|
|
} catch (error: any) {
|
|
lastError = error instanceof Error ? error : new Error(String(error));
|
|
logError(`[AIService] ${engine} failed:`, lastError.message);
|
|
}
|
|
}
|
|
|
|
// Both engines exhausted. Surface a result with empty=true so the
|
|
// caller (e.g. Telegram handler) can produce a user-visible reply
|
|
// instead of swallowing the failure.
|
|
if (lastEmptyEngine) {
|
|
return { content: '', engine: lastEmptyEngine, model, empty: true };
|
|
}
|
|
throw lastError || new Error('All AI engines failed.');
|
|
}
|
|
}
|
|
|
|
/**
|
|
* [코어 수렴] 인스턴스 없이 코어 LLM 경로를 쓰는 모듈 함수용 헬퍼.
|
|
* AIService 는 무상태(설정을 매 호출 읽음)라 안전하다. 엔진 폴백·타임아웃·abort·
|
|
* 로깅·빈응답 소프트실패가 모두 이 한 경로로 통일된다 — 자체 fetch 금지.
|
|
*/
|
|
export function coreChat(req: AIChatRequest): Promise<AIChatResult> {
|
|
return new AIService().chat(req);
|
|
}
|
|
|
|
/**
|
|
* BrainService: 지식 베이스 파일 시스템 저장 및 관리 구현체
|
|
*/
|
|
export class BrainService implements IBrainService {
|
|
public async inject(title: string, markdown: string): Promise<string> {
|
|
const brainDir = _getBrainDir();
|
|
if (!fs.existsSync(brainDir)) {
|
|
fs.mkdirSync(brainDir, { recursive: true });
|
|
}
|
|
|
|
const today = new Date().toISOString().split('T')[0];
|
|
const datePath = path.join(brainDir, '00_Raw', today);
|
|
if (!fs.existsSync(datePath)) {
|
|
fs.mkdirSync(datePath, { recursive: true });
|
|
}
|
|
|
|
const safeTitle = title.replace(/[^a-zA-Z0-9가-힣]/gi, '_');
|
|
const filePath = path.join(datePath, `${safeTitle}.md`);
|
|
fs.writeFileSync(filePath, markdown, 'utf-8');
|
|
|
|
return filePath;
|
|
}
|
|
}
|