fix(suggestion): paste after the shortcut keys are released and keep typing detection stable

Ctrl+Alt+Enter pasted while Ctrl+Alt were still down, so the target app got
Ctrl+Alt+V; accepting now closes the panel and waits for the modifiers to be
released. Candidates are accepted on pointer press because the list is
redrawn as new candidates stream in, which swallowed clicks.

The typing gate identified the focused field by its bounds, so chat boxes
that grow while typing looked like a new field on every keystroke and were
reported as "not typing". Fields are now keyed by window, control type and
name, and a mouse click re-baselines the text instead. The decision log
includes both gate values.

The live-caption model selector moves to the caption section of the General
tab, next to the other caption settings.
This commit is contained in:
Yun Chan 2026-09-24 22:01:10 +09:00
parent 4b0f685941
commit da0da98285
8 changed files with 144 additions and 55 deletions

View file

@ -35,7 +35,7 @@ import { useI18n, LOCALE_META } from '@d3ro/i18n'
import type { Locale } from '@d3ro/i18n'
import type { KeyBindingActionGroup } from '@d3ro/core/keybinding'
import { auditKeyBindingMap, KEYBINDING_ACTIONS } from '@d3ro/core/keybinding'
import type { ThemeMode, AppConfig, AudioDevice, LLMModel, LLMStatus } from '@d3ro/core/types'
import type { ThemeMode, AppConfig, AudioDevice, LLMModel, LLMStatus, STTModel } from '@d3ro/core/types'
import { useKeyBindingMap } from '../hooks/useKeyBindingMap'
interface SettingsModalProps {
@ -64,6 +64,9 @@ const ACTION_GROUP_LABEL_KEYS: Readonly<Record<KeyBindingActionGroup, string>> =
const ACTION_GROUP_ORDER: readonly KeyBindingActionGroup[] = ['voice', 'window', 'input']
/** 자막 모델 선택지 "받아쓰기와 같게" — 설정에는 null 로 저장한다 */
const SAME_AS_DICTATION = '__same__'
export function SettingsModal({ open, initialTab = 0, onClose }: SettingsModalProps): React.ReactElement {
const { t, locale, setLocale } = useI18n()
const [activeTab, setActiveTab] = useState(initialTab)
@ -108,6 +111,15 @@ export function SettingsModal({ open, initialTab = 0, onClose }: SettingsModalPr
const [micTesting, setMicTesting] = useState(false)
const [micLevel, setMicLevel] = useState(0)
// 실시간 자막 모델 선택지 — 받아 둔 로컬 Whisper 모델만
const [downloadedSttModels, setDownloadedSttModels] = useState<STTModel[]>([])
useEffect(() => {
if (!open) return
window.electronAPI.stt.getModels().then((res) => {
if (res.success && res.data) setDownloadedSttModels(res.data.filter((m) => m.downloaded))
})
}, [open])
// Ollama 상태
const [ollamaStatus, setOllamaStatus] = useState<LLMStatus | null>(null)
const [ollamaModels, setOllamaModels] = useState<LLMModel[]>([])
@ -481,6 +493,27 @@ export function SettingsModal({ open, initialTab = 0, onClose }: SettingsModalPr
</Select>
</FormControl>
<FormControl size="small">
<InputLabel>{t('settings.captionModel')}</InputLabel>
<Select
label={t('settings.captionModel')}
value={config.captionSttModelId ?? SAME_AS_DICTATION}
onChange={(e) =>
updateConfig('captionSttModelId', e.target.value === SAME_AS_DICTATION ? null : e.target.value)
}
>
<MenuItem value={SAME_AS_DICTATION}>{t('settings.captionModel.same')}</MenuItem>
{downloadedSttModels.map((m) => (
<MenuItem key={m.id} value={m.id}>
{m.name}
</MenuItem>
))}
</Select>
<Typography sx={{ mt: 0.5, fontSize: d3roTypo.meta.size, color: d3roPalette.text.secondary }}>
{t('settings.captionModel.desc')}
</Typography>
</FormControl>
<FormControlLabel
sx={{ mr: 0, alignItems: 'flex-start' }}
control={