Files
connectai/src/core/services.ts
T
koriwebandClaude Fable 5 47b3b9f93a v2.2.299~307: 아키텍처 수렴 + 채팅 정리 + 벤치마킹 강화 + Claude 구독 엔진
- v2.2.299 아키텍처 수렴: coreChat 통일(엔진 휴리스틱 3벌 제거), 기업 모드
  검색 오케스트레이터 승격, lib/execUtil(실행 래퍼 6곳·Python 탐지 3벌 단일화),
  lib/kstSchedule(워처 4개 nowInKst 통합), estimateTokens 통합, 설정 접근 규칙 명문화
- v2.2.300 채팅 화면 정리: LiveReasoningFilter(스트리밍 중 <think>/Harmony 추론
  토큰 단위 차단), 확신도·검토요청 footer 기본 숨김(계산·Reflection 은 유지)
- v2.2.301 문맥·의도 이해: [답변 전 이해 원칙] 상시 주입, 워크플로우 의도 브리핑,
  Report QA 루프(규칙 레지스트리+실측치+회귀 게이트, 블로그_v3 개념 이식)
- v2.2.302 /benchmark 비즈니스 렌즈(가격·수익·운영)+빌드 프롬프트 모드+QA 연계
- v2.2.303 handoff 모드(측정치 무손실 인수인계 문서)+/claude(Claude Code 터미널 위임)
- v2.2.304 이식성: 지식 경로 두뇌-상대 규약(pickWikiDir 상대 해석), 이사 체크리스트
- v2.2.305 Claude 구독 엔진: claude: 프로바이더(CLI 위임, 모델 드롭다운 자동 노출,
  coreChat 지원 — 워크플로우·QA도 구독 모델 가능)
- v2.2.306 Tone Guard: AI 상투어 금지 레지스트리(상담사 화법 실사례 8종+대조 예시)
- v2.2.307 /benchmark 레이아웃 골격(sectionRoles 결정론 분석, 롤링 배너 즉답),
  파트별 실패 격리, 합성 타임아웃 120→300초

검증: tsc 무오류 + jest 888 통과 + esbuild 정상

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-11 21:02:15 +09:00

219 lines
9.5 KiB
TypeScript

import * as fs from 'fs';
import * as path from 'path';
import { getConfig } from '../config';
import { buildApiUrl, logError, logInfo, resolveEngine, summarizeText, _getBrainDir } from '../utils';
/**
* IAIService: AI 모델 호출에 대한 인터페이스.
*
* `call(prompt)` 는 plain user 메시지 1개만 보내는 legacy shortcut이고,
* `chat({ system, user })` 는 role-aware 호출이다. Telegram 핸들러처럼
* 모델을 grounding 해야 하는 경로에서는 system을 반드시 채워야 한다 —
* gemma 같은 작은 모델은 system이 없으면 짧은/모호한 입력에 대해
* "시는 못 써드려요" 같은 환각 거절을 하는 경향이 있다.
*/
export interface IAIService {
call(prompt: string): Promise<string>;
chat(req: AIChatRequest): Promise<AIChatResult>;
}
export interface AIChatRequest {
/** Optional system prompt. Strongly recommended for short / ambiguous user inputs. */
system?: string;
/** Required. The user message. */
user: string;
/** Optional override (default = config.defaultModel). */
model?: string;
/** Optional override (default = config.timeout). */
timeoutMs?: number;
/** 샘플링 온도 (기본 0.7). 판정·재순위 등 결정적 작업은 0.0~0.2 권장. */
temperature?: number;
/** 출력 토큰 상한 (ollama num_predict / lmstudio max_tokens). 미지정 시 서버 기본. */
maxTokens?: number;
/** [ollama 전용] 컨텍스트 창 크기(num_ctx). */
numCtx?: number;
/**
* 외부 abort signal. fetch 가 받는 signal 과 OR 로 결합되어, 사용자가 회사 모드
* 도중 Stop 을 누르면 진행 중인 generation 이 즉시 중단된다. 없으면 timeout 만
* 적용. dispatcher 같은 긴 multi-turn 경로에서 반드시 전달할 것.
*/
signal?: AbortSignal;
}
export interface AIChatResult {
content: string;
/** Engine that actually returned the content. */
engine: 'lmstudio' | 'ollama' | 'claude-code';
model: string;
/** True iff content came back empty after all retries. Caller decides UX. */
empty: boolean;
}
/**
* IBrainService: 지식 베이스(Brain) 조작에 대한 인터페이스
*/
export interface IBrainService {
inject(title: string, markdown: string): Promise<string>;
}
/**
* AIService: Ollama 및 LM Studio 폴백 로직을 포함한 AI 호출 구현체.
*
* Behavior:
* 1. Try the user-configured engine first; on transport / 5xx / empty response,
* fall through to the other engine.
* 2. Empty responses are treated as a soft failure: we log + retry the other
* engine before giving up. Pure exceptions (network blip) trigger the same
* fallback path.
* 3. The legacy `call(prompt)` is preserved as a thin wrapper around `chat()`
* for callers that don't have a system prompt — but new code should pass
* a system prompt explicitly.
*/
export class AIService implements IAIService {
public async call(prompt: string): Promise<string> {
const result = await this.chat({ user: prompt });
return result.content;
}
public async chat(req: AIChatRequest): Promise<AIChatResult> {
const config = getConfig();
const model = (req.model || config.defaultModel || '').trim() || 'gemma4:e2b';
const timeoutMs = req.timeoutMs ?? config.timeout;
// [v2.2.305] Claude 구독 모델 ('claude:sonnet' 등) — 로컬 엔진 대신 Claude Code CLI 위임.
// 워크플로우·Report QA·리랭크 등 coreChat 소비자 전부가 구독 모델로 동작 가능해진다.
// (다른 클라우드 prefix 는 스트리밍 전용 경로만 지원 — 여기 오면 아래 로컬 시도가 실패로 드러남)
if (model.startsWith('claude:')) {
const { runClaudeCode } = await import('../features/providers/claudeCode');
const messages: Array<{ role: 'system' | 'user' | 'assistant'; content: string }> = [];
if (req.system && req.system.trim()) messages.push({ role: 'system', content: req.system });
messages.push({ role: 'user', content: req.user });
const r = await runClaudeCode({
messages,
model: model.slice('claude:'.length),
signal: req.signal,
timeoutMs: Math.max(timeoutMs, 120_000),
});
return { content: r.text, engine: 'claude-code', model, empty: !r.text.trim() };
}
const primaryEngine = resolveEngine(config.ollamaUrl);
const engines = primaryEngine === 'lmstudio'
? ['lmstudio', 'ollama'] as const
: ['ollama', 'lmstudio'] as const;
const messages: Array<{ role: 'system' | 'user' | 'assistant'; content: string }> = [];
if (req.system && req.system.trim()) {
messages.push({ role: 'system', content: req.system });
}
messages.push({ role: 'user', content: req.user });
let lastError: Error | null = null;
let lastEmptyEngine: typeof engines[number] | null = null;
for (const engine of engines) {
const apiUrl = buildApiUrl(config.ollamaUrl, engine, 'chat');
const temperature = req.temperature ?? 0.7;
const payload = {
model,
messages,
stream: false,
...(engine === 'ollama'
? { options: {
temperature,
...(req.maxTokens != null ? { num_predict: req.maxTokens } : {}),
...(req.numCtx != null ? { num_ctx: req.numCtx } : {}),
} }
: {
temperature,
...(req.maxTokens != null ? { max_tokens: req.maxTokens } : {}),
}),
};
try {
logInfo('[AIService] Request started.', {
engine, apiUrl, model,
hasSystem: !!req.system, userChars: req.user.length,
});
// timeout signal + 외부 abort signal 결합. 외부 signal 이 fire 되면
// 진행 중인 fetch 가 즉시 중단되어 사용자 Stop 이 LLM generation 중에도 효과.
const timeoutSignal = AbortSignal.timeout(timeoutMs);
const combinedSignal = req.signal
? AbortSignal.any([req.signal, timeoutSignal])
: timeoutSignal;
const res = await fetch(apiUrl, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(payload),
signal: combinedSignal,
});
const rawText = await res.text();
if (!res.ok) {
lastError = new Error(`AI call failed: ${res.status} ${summarizeText(rawText, 250)}`);
logError(`[AIService] ${engine} HTTP ${res.status}`, { body: summarizeText(rawText, 250) });
continue;
}
const data = rawText ? JSON.parse(rawText) as any : {};
const content = engine === 'lmstudio'
? (data.choices?.[0]?.message?.content || '')
: (data.message?.content || data.response || '');
if (!content || !content.trim()) {
// Treat empty as soft failure so the other engine gets a chance.
lastEmptyEngine = engine;
lastError = new Error(`AI engine '${engine}' returned an empty response.`);
logError(`[AIService] ${engine} empty response — falling through.`, { model });
continue;
}
return { content, engine, model, empty: false };
} catch (error: any) {
lastError = error instanceof Error ? error : new Error(String(error));
logError(`[AIService] ${engine} failed:`, lastError.message);
}
}
// Both engines exhausted. Surface a result with empty=true so the
// caller (e.g. Telegram handler) can produce a user-visible reply
// instead of swallowing the failure.
if (lastEmptyEngine) {
return { content: '', engine: lastEmptyEngine, model, empty: true };
}
throw lastError || new Error('All AI engines failed.');
}
}
/**
* [코어 수렴] 인스턴스 없이 코어 LLM 경로를 쓰는 모듈 함수용 헬퍼.
* AIService 는 무상태(설정을 매 호출 읽음)라 안전하다. 엔진 폴백·타임아웃·abort·
* 로깅·빈응답 소프트실패가 모두 이 한 경로로 통일된다 — 자체 fetch 금지.
*/
export function coreChat(req: AIChatRequest): Promise<AIChatResult> {
return new AIService().chat(req);
}
/**
* BrainService: 지식 베이스 파일 시스템 저장 및 관리 구현체
*/
export class BrainService implements IBrainService {
public async inject(title: string, markdown: string): Promise<string> {
const brainDir = _getBrainDir();
if (!fs.existsSync(brainDir)) {
fs.mkdirSync(brainDir, { recursive: true });
}
const today = new Date().toISOString().split('T')[0];
const datePath = path.join(brainDir, '00_Raw', today);
if (!fs.existsSync(datePath)) {
fs.mkdirSync(datePath, { recursive: true });
}
const safeTitle = title.replace(/[^a-zA-Z0-9가-힣]/gi, '_');
const filePath = path.join(datePath, `${safeTitle}.md`);
fs.writeFileSync(filePath, markdown, 'utf-8');
return filePath;
}
}