Compare commits

..

11 Commits

  1. BIN
      vite/public/assets/moty/q1.mp3
  2. BIN
      vite/public/assets/moty/q2-1.mp3
  3. BIN
      vite/public/assets/moty/q2.mp3
  4. BIN
      vite/public/assets/moty/q3.mp3
  5. BIN
      vite/public/assets/moty/q5.mp3
  6. BIN
      vite/public/assets/moty/q7-1-1.mp3
  7. BIN
      vite/public/assets/moty/q7-1-2.mp3
  8. BIN
      vite/public/assets/moty/q7-1.mp3
  9. BIN
      vite/public/assets/moty/q7-2.mp3
  10. BIN
      vite/public/assets/moty/q7.mp3
  11. BIN
      vite/public/assets/moty/q8-1.mp3
  12. BIN
      vite/public/assets/moty/q8.mp3
  13. 141
      vite/public/cuelist_moty.json
  14. 19
      vite/public/default.json
  15. 3
      vite/src/App.jsx
  16. 24
      vite/src/comps/numpad.jsx
  17. 4
      vite/src/main.jsx
  18. 105
      vite/src/pages/flow_free.jsx
  19. 1162
      vite/src/pages/flow_moty.jsx
  20. 4
      vite/src/pages/settings.jsx
  21. 37
      vite/src/util/multipart.js
  22. 8
      vite/src/util/osc.js
  23. 274
      vite/src/util/stt.js
  24. 50
      vite/src/util/system_prompt.js
  25. 384
      vite/src/util/useSpeechInput.jsx
  26. 3
      vite/src/util/useUser.jsx

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

@ -0,0 +1,141 @@
{
"cuelist": [
{
"id": 1,
"name": "Q1",
"type": "phone",
"description": "preset announce",
"audioFile": "assets/moty/q1.mp3",
"layer":"announce",
"loop": true,
"status":"reset",
"fadeout": true,
"nextcue": 2,
"callback":"numpad",
"numpad_type":"enter",
"input_time": 0,
"soundcue":"Q1"
},
{
"id": 2,
"name": "Q2",
"type": "phone",
"description": "引導輸入電話號碼",
"auto": true,
"audioFile": "assets/moty/q2.mp3",
"nextcue": 2.1,
"callback":"numpad",
"numpad_type":"phonenum",
"input_time": 18000 ,
"status":"intro",
"hint":"請輸入你的電話號碼\n按下#鍵,完成輸入\n按下*鍵,重新輸入",
"hint_time": 6900,
"soundcue":"Q2"
},
{
"id":2.1,
"name": "Q2.1",
"type": "phone",
"description": "撥打音效",
"auto": true,
"audioFile": "assets/moty/q2-1.mp3",
"nextcue": 3,
"hint":"輸入完成",
"hint_time":100
},
{
"id": 3,
"name": "Q3",
"type": "phone",
"description": "引導生圖",
"auto": true,
"audioFile": "assets/moty/q3.mp3",
"nextcue": 4,
"hint":"你想像中的美好未來\n長什麼樣子?",
"hint_time":3000
},
{
"id": 4,
"name": "Q4",
"type": "chat",
"description": "chat",
"auto": true,
"nextcue": 5,
"duration": 90,
"status":"go",
"chatInterval":20,
"soundcue":"Q3"
},
{
"id": 5,
"name": "Q5",
"type": "phone",
"description": "提取完成",
"auto": true,
"audioFile": "assets/moty/q5.mp3",
"nextcue": 6
},
{
"id": 6,
"name": "Q6",
"type": "user_input",
"description": "call",
"duration": 20,
"auto": true,
"nextcue": 6.1
},
{
"id":6.1,
"name":"Q6.1",
"type":"export",
"auto":true,
"description":"export",
"callback":"exportFile",
"nextcue":7,
"soundcue":"Q1",
"duration":1
},
{
"id": 7,
"name": "Q7",
"type": "phone",
"description": "Ending",
"auto": true,
"audioFile": "assets/moty/q7-1-1.mp3",
"nextcue": 7.1,
"status":"intro",
"hint":"裝置使用完畢\n祝福你抵達你所期望的未來",
"hint_time":3500
},
{
"id": 7.1,
"name": "Q7.1",
"type": "phone",
"description": "Ad",
"auto": true,
"audioFile": "assets/moty/q7-1-2.mp3",
"nextcue": 8,
"status":"ad"
},
{
"id": 8,
"name": "Q8",
"type": "end",
"description": "QRcode",
"status":"intro",
"audioFile": "assets/moty/q8.mp3",
"auto": true,
"nextcue": 8.1
},{
"id": 8.1,
"name": "Q8.1",
"type": "end",
"description": "QRcode",
"status":"end",
"audioFile": "assets/moty/q8-1.mp3",
"auto": true,
"nextcue": 1
}
]
}

@ -1,17 +1,10 @@
{
"system_prompt": "你是一位具有同理心的 AI 助理,透過溫柔的語氣,引導使用者回想並表達一段內心的遺憾或未竟之事,從場景、人物、動作到感受。\n你的任務是協助使用者逐步揭開這段記憶的情緒層次,並在每一階段輸出一句英文圖像生成簡短的 Prompt,讓這段過往漸漸具象為一幅畫面。\n以溫柔、自然、短問句,台灣語境的繁體中文引導,每次只回應一個問題。",
"voice": "nova",
"system_prompt": "你是一位能產生圖像的 AI 助理,專長是以幽默、俏皮式的提問,帶領使用者輕鬆玩味想像中的美好未來。你的任務是逐步引導使用者自由、充滿創意地描述那幅美好畫面,並將他們的描述化為圖像生成提示詞,讓畫面從遠景漸漸聚焦到未來的自己與場景。生成的圖像應避免僅有人臉正面特寫,而是包含背景元素,以側面或背影呈現人物。每次以台灣繁體中文,用一個輕鬆、有趣的短問句提問。若使用者回答與「美好未來」無關,試著調皮地提醒他們回到「遊戲」畫面中,但不強迫。AI助理不得自行創造或描述場景內容,所有圖像提示詞都必須嚴格基於使用者的回答。",
"voice": "onyx",
"voice_prompt": "Speak as a gentle, grounded Taiwanese narrator with a warm local accent. Use a soft, soothing, and deeply compassionate tone, with slow and deliberate pacing. Pause often between phrases and within sentences, as if listening and breathing with the listener. Convey patient attentiveness—not rushing to comfort, but quietly staying present. Pronounce each word clearly, softly, and slightly slowly, letting every word land with warmth and care.",
"summary_prompt": "請將這段口白的核心情感,轉化為一句不超過 50 字的抽象化描述。這句話應保有距離感,只勾勒出情感的輪廓,同時暗示著一種持續前行、未完待續的狀態,語氣平實。",
"summary_prompt": "請將這段口白的核心感受,轉化為一句對畫面中「未來的自己」或「未來場景」的告白或陳述,不超過 50 個字。這句話是第一人稱對第二人稱說的話,語氣要能讓人感受到一種淡淡的、未完待續的心情,同時帶點希望。請使用台灣語境的繁體中文。",
"speech_idle_time": "4000",
"sd_prompt_prefix": "a luminous impression of a {{",
"sd_prompt_suffix": "}}, a whispered Taiwanese memory, an iridescent wash of colors, shimmering light, ethereal, optimistic tone, fading contours, a gentle touch, dreamlike clarity, (bright ambient light), (subtle lens flare), (high key lighting), peaceful and hopeful.",
"stt_mode": "realtime",
"stt_model": "gpt-4o-transcribe",
"stt_language": "zh-tw",
"stt_prompt": "以台灣繁體中文輸出,保留口語停頓。",
"vad_threshold": "0.5",
"vad_silence_ms": "500",
"stt_max_segment_ms": "8000",
"speech_flush_lead": "3000"
}
"sd_prompt_suffix": "}}, an anticipated vision, iridescent, soft blur, bokeh, peaceful, dreamlike, figure, back or side of the figure",
"welcome_prompt": `使`
}

@ -10,7 +10,8 @@ function App() {
<div className='w-full flex flex-row gap-2 justify-center py-2 px-8 *:bg-pink-200 *:px-2'>
{/* <a href="/">Conversation</a> */}
{/* <a href="/flow">Flow</a> */}
<a href="/free-flow">Free Flow</a>
<a href="/moty">Moty</a>
{/* <a href="/free-flow">Free Flow</a> */}
<a href="/settings">Settings</a>
</div>
)

@ -6,6 +6,8 @@ export const NUMPAD_TYPE={
USERID: 'userid',
PASSWORD: 'password',
CHOICE: 'choice',
PHONE:'phonenum',
ENTER:'enter',
}
const KEY_ENTER='d';
@ -21,7 +23,7 @@ const TMP_MAP_KEY={
7:9,*/
}
export default function NumPad({onSend, disabled, type, clientId}){
export default function NumPad({onSend, disabled, type, clientId, onUserInput, ...props}){
const [input, _setInput]=useState();
const refInput=useRef();
@ -52,11 +54,19 @@ export default function NumPad({onSend, disabled, type, clientId}){
console.log(e.key);
onUserInput();
if(disabled) return; // Ignore key events if disabled
if(e.key===KEY_ENTER){
if(refType.current==NUMPAD_TYPE.ENTER){
refLatestInput.current=KEY_ENTER;
refAudio.current[KEY_ENTER]?.play();
setInput(()=>'');
return;
}
if(refInput.current && refInput.current.length>0){
const num=parseInt(refInput.current);
@ -92,6 +102,16 @@ export default function NumPad({onSend, disabled, type, clientId}){
refAudio.current['error']?.play();
}
break;
case NUMPAD_TYPE.PHONE:
if(refInput.current.length==10){
// onSend(refInput.current);
refLatestInput.current=refInput.current;
refAudio.current[KEY_ENTER]?.play();
}else{
refAudio.current['error']?.play();
}
break;
}
@ -177,7 +197,7 @@ export default function NumPad({onSend, disabled, type, clientId}){
useEffect(()=>{
if(disabled) return;
// if(disabled) return;
window.onkeydown=onkeydown;

@ -6,6 +6,7 @@ import './index.css'
import App from './App.jsx'
import { Settings } from './pages/settings.jsx';
import { Flow } from './pages/flow.jsx';
import { FlowMoty } from './pages/flow_moty.jsx';
import { Conversation } from './pages/conversation.jsx';
import { ChatProvider } from './util/useChat.jsx';
import { DataProvider } from './util/useData.jsx';
@ -20,10 +21,11 @@ createRoot(document.getElementById('root')).render(
<BrowserRouter>
<App />
<Routes>
<Route path="/" element={<FreeFlow />} />
<Route path="/" element={<FlowMoty />} />
<Route path="/flow" element={<Flow />} />
<Route path="/free-flow" element={<FreeFlow />} />
<Route path="/settings" element={<Settings />} />
<Route path="/moty" element={<FlowMoty />} />
</Routes>
</BrowserRouter>
</ChatProvider>

@ -1,6 +1,6 @@
import { invoke } from '@tauri-apps/api/core';
import { useEffect, useRef, useState } from "react";
import useSpeechInput from '../util/useSpeechInput';
import SpeechRecognition, { useSpeechRecognition } from 'react-speech-recognition';
import { Countdown } from "../comps/timer";
@ -18,7 +18,6 @@ import { useUser } from "../util/useUser";
const CUELIST_FILE = 'cuelist_1009.json';
const AUDIO_FADE_TIME=3000; // in ms
const BLUR=false;
const EmojiType={
phone: '📞',
@ -103,14 +102,9 @@ export function FreeFlow(){
finalTranscript,
listening,
resetTranscript,
browserSupportsSpeechRecognition,
isMicrophoneAvailable,
start: startRecognition,
stop: stopRecognition,
flush: flushSpeech,
}=useSpeechInput(data, {
onSpeechStart: handleSpeechStart,
onSpeechStop: handleSpeechStop,
});
}=useSpeechRecognition();
function resetData() {
@ -538,24 +532,6 @@ export function FreeFlow(){
sendOsc(OSC_ADDRESS.SPEECH, 'stop');
refSpeaking.current=false;
}
// VAD realtime
// delta VAD commit transcript
function handleSpeechStart(){
sendSpeechStart();
if(refPauseTimer.current) clearTimeout(refPauseTimer.current);
}
// VAD
function handleSpeechStop(){
if(refCurrentCue.current?.type!='chat') return;
sendSpeechEnd();
if(chatStatus!=ChatStatus.User) return;
setPauseTimer();
}
function onCueEnd() {
refTimer.current?.stop(); // Stop the timer when cue ends
@ -656,25 +632,11 @@ export function FreeFlow(){
let timeleft=refCurrentCue.current?.chatInterval || 0;
const endTime=new Date().getTime()+timeleft*1000;
const flushLead=Number(data?.speech_flush_lead) || 2000;
let flushed=false;
function tick(){
const now=new Date().getTime();
const timeLeft=endTime-now;
// commit 1.2
// processSpeech textarea
//
// lead 0.9 timeleft tick Math.floor
// endTime 0.8-0.9 lead
// = speech_flush_lead - 900ms
if(!flushed && timeLeft<=flushLead){
flushed=true;
console.log('~~~ chat timer near end, flush speech tail');
flushSpeech();
}
if(timeleft<=0){
console.log('~~~ chat timer ended, process speech');
clearInterval(refChatTimer.current);
@ -783,16 +745,18 @@ export function FreeFlow(){
function onSpeechEnd(){
if(currentCue?.type!='chat') return; // Only process if current cue is user input
if(chatStatus!=ChatStatus.User) return; // Only process if chat status is User
// 12
// speech_idle_time 4
// handleSpeechStop
if(refSpeaking.current) return;
sendSpeechEnd();
console.log('~~~ on speech end, start pause timer',data.speech_idle_time);
// refSpeechPaused.current=true;
setPauseTimer();
}
function processSpeech(){
@ -857,30 +821,46 @@ export function FreeFlow(){
},[finalTranscript]);
function blurText(text) {
if(!BLUR) return text;
function startRecognition() {
SpeechRecognition.startListening({ continuous: true, language: 'zh-TW' }).then(() => {
console.log("Speech recognition started.");
}).catch(error => {
console.error("Error starting speech recognition:", error);
});
}
function blurText(text) {
if(!text) return '';
return text.replace(/./g, '*');
}
useEffect(()=>{
if(audioInput && isMicrophoneAvailable) {
startRecognition();
const recognition= SpeechRecognition.getRecognition();
recognition.onspeechstart=(e)=>{
console.log('Speech start:', e);
sendSpeechStart();
};
// recognition.onspeechend=(e)=>{
// console.log('Speech end:', e);
// startRecognition();
// };
// transcript cue Realtime
// chatStatus
// isMicrophoneAvailable start false
// stop() start()
// log chat cue
const wantsAudio = audioInput
&& (currentCue?.type=='chat' || currentCue?.type=='user_input');
if(wantsAudio) {
startRecognition();
}else{
stopRecognition();
console.log('Stopping speech recognition...');
SpeechRecognition.stopListening();
}
},[audioInput, currentCue, startRecognition, stopRecognition]);
},[audioInput]);
useEffect(()=>{
@ -1108,7 +1088,7 @@ export function FreeFlow(){
</div>
</section>
<section className="flex-1 self-stretch overflow-y-auto flex flex-col justify-end gap-2 ">
<div ref={refContainer} className="flex-1 flex flex-col overflow-y-auto gap-2" >
<div ref={refContainer} className="flex-1 flex flex-col overflow-y-auto gap-2 blur-sm" >
{history?.map((msg, index) => (
<div key={index} className={`w-5/6 ${msg.role=='user'? 'self-end':''}`}>
<div className={`${msg.role=='user'? 'bg-green-300':'bg-pink-300'} px-2`}>{blurText(msg.content)}</div>
@ -1118,7 +1098,7 @@ export function FreeFlow(){
{summary && <div className="w-full self-center bg-blue-200 px-2">{summary}</div>}
</div>
<textarea ref={refInput} name="message" rows={2}
className={`w-full border-1 resize-none p-2 disabled:bg-gray-500`}
className={`w-full border-1 resize-none p-2 disabled:bg-gray-500 blur-sm`}
disabled={chatStatus!=ChatStatus.User && chatStatus!=ChatStatus.Message}></textarea>
<div className="flex flex-row justify-end gap-2 flex-wrap">
<span className="flex flex-row gap-1">
@ -1129,9 +1109,6 @@ export function FreeFlow(){
<label>audio_input</label>
<input type='checkbox' checked={audioInput} onChange={(e) => setAudioInput(e.target.checked)} />
</span>
{!isMicrophoneAvailable && (
<div className="rounded-2xl bg-red-400 self-end px-4 tracking-widest">mic_unavailable</div>
)}
<span className="flex flex-row gap-1">
<label>auto_send</label>
<input type='checkbox' checked={autoSend} onChange={(e) => setAutoSend(e.target.checked)} />

File diff suppressed because it is too large Load Diff

@ -34,8 +34,8 @@ export function Settings(){
{data && Object.entries(data).map(([key, value], index) => (
<div key={index} className='flex flex-col gap-1 flex-1'>
<label className='bg-gray-200 self-start px-2'>{key}</label>
{["speech_idle_time","vad_threshold","vad_silence_ms","stt_max_segment_ms","speech_flush_lead"].includes(key) ? (
<input name={key} type='number' step='any' defaultValue={value} className='border'></input>
{key=="speech_idle_time" ? (
<input name={key} type='number' defaultValue={value} className='border'></input>
):(
<textarea name={key} defaultValue={value} className='border flex-1'></textarea>
)}

@ -1,37 +0,0 @@
// tauri 的 fetch 不保證吃得下 FormData,multipart 自己組成 bytes 最穩。
// 這個檔案刻意不 import 任何 tauri 模組,才能在 node 裡直接跑測試。
/**
* @param {Record<string, string|undefined>} fields 文字欄位undefined/空字串會被略過
* @param {{filename: string, type: string, bytes: Uint8Array}} file
* @returns {{boundary: string, body: Uint8Array}}
*/
export function buildMultipart(fields, file) {
const boundary = `----tipsy${Math.random().toString(16).slice(2)}`;
const encoder = new TextEncoder();
const parts = [];
for (const [name, value] of Object.entries(fields)) {
if (value === undefined || value === null || value === '') continue;
parts.push(encoder.encode(
`--${boundary}\r\nContent-Disposition: form-data; name="${name}"\r\n\r\n${value}\r\n`
));
}
parts.push(encoder.encode(
`--${boundary}\r\nContent-Disposition: form-data; name="file"; filename="${file.filename}"\r\n` +
`Content-Type: ${file.type}\r\n\r\n`
));
parts.push(file.bytes);
parts.push(encoder.encode(`\r\n--${boundary}--\r\n`));
const size = parts.reduce((total, part) => total + part.byteLength, 0);
const body = new Uint8Array(size);
let offset = 0;
for (const part of parts) {
body.set(part, offset);
offset += part.byteLength;
}
return { boundary, body };
}

@ -30,10 +30,14 @@ export const OSC_ADDRESS={
SPEECH_PAUSE:'/speech_pause',
TEST_EXPORT:'/test_export',
SCS_PLAY_CUE:'/cue/go',
SCS_STOP_CUE:'/cue/stop',
SCS_STOP_ALL:'/ctrl/stopall',
}
export async function sendOsc(key, message){
export async function sendOsc(key, message, port='9000') {
if(message === undefined || message === null) {
console.warn('sendOsc: message is empty, skipping');
@ -49,7 +53,7 @@ export async function sendOsc(key, message){
key: key,
message: message.toString(),
host:`0.0.0.0:0`,
target: '127.0.0.1:9000',
target: `127.0.0.1:${port}`,
});
}catch (error){
console.error('Error sending OSC message:', error);

@ -1,274 +0,0 @@
import { fetch } from '@tauri-apps/plugin-http';
import { invoke } from '@tauri-apps/api/core';
import { buildMultipart } from './multipart';
// OpenAI speech-to-text 的傳輸層。兩條路:
// - realtime: WebSocket 長連線,邊說邊吐 delta,斷句由 server VAD 決定
// - batch: 錄成一段 webm 再整段丟 /v1/audio/transcriptions
// 兩者都由 useSpeechInput 包成跟舊的 useSpeechRecognition 同形的介面。
const REALTIME_URL = 'wss://api.openai.com/v1/realtime?intent=transcription';
const SAMPLE_RATE = 24000;
// AudioWorklet 只能用 URL 載入,走 Blob 比丟進 public/ 少一個部署時會忘記的檔案。
// AudioContext 開在 24kHz,重採樣交給瀏覽器,這裡只負責 Float32 -> Int16。
const PCM_WORKLET_SRC = `
class PCMWorklet extends AudioWorkletProcessor {
process(inputs) {
const input = inputs[0]?.[0];
if (!input) return true;
const pcm = new Int16Array(input.length);
for (let i = 0; i < input.length; i++) {
const s = Math.max(-1, Math.min(1, input[i]));
pcm[i] = s < 0 ? s * 0x8000 : s * 0x7fff;
}
this.port.postMessage(pcm.buffer, [pcm.buffer]);
return true;
}
}
registerProcessor('pcm-worklet', PCMWorklet);
`;
async function getOpenAIToken() {
return invoke('get_env', { name: 'OPENAI_API_KEY' });
}
function encodeBase64(buffer) {
const bytes = new Uint8Array(buffer);
let binary = '';
// 分段避免 String.fromCharCode 參數過多爆堆疊
for (let i = 0; i < bytes.length; i += 0x8000) {
binary += String.fromCharCode(...bytes.subarray(i, i + 0x8000));
}
return btoa(binary);
}
function sttModel(data) {
return data?.stt_model?.trim() || 'gpt-4o-transcribe';
}
// gpt-live-transcribe 自己決定斷句,帶 turn_detection 會被 400 擋掉。
function supportsTurnDetection(model) {
return model !== 'gpt-live-transcribe';
}
// whisper-1 帶 chunking_strategy 會 400:"chunking_strategy is not supported with this model"。
function supportsChunkingStrategy(model) {
return model !== 'whisper-1';
}
// gpt-4o-transcribe / gpt-live-transcribe 收得下 zh-tw 這種區域碼,
// whisper-1 只收 ISO-639-1,給它 zh-tw 會 400,得把後綴削掉靠 prompt 顧繁體。
function sttLanguage(data, model) {
const language = data?.stt_language?.trim() || 'zh-tw';
return model === 'whisper-1' ? language.split('-')[0] : language;
}
function transcriptionSession(data) {
const model = sttModel(data);
const input = {
format: { type: 'audio/pcm', rate: SAMPLE_RATE },
transcription: {
model,
language: sttLanguage(data, model),
},
noise_reduction: { type: 'near_field' },
};
const prompt = data?.stt_prompt?.trim();
if (prompt) input.transcription.prompt = prompt;
if (supportsTurnDetection(model)) {
input.turn_detection = {
type: 'server_vad',
threshold: Number(data?.vad_threshold) || 0.5,
silence_duration_ms: Number(data?.vad_silence_ms) || 500,
prefix_padding_ms: 300,
};
}
return { type: 'transcription', audio: { input } };
}
async function createEphemeralToken(data) {
const token = await getOpenAIToken();
const response = await fetch('https://api.openai.com/v1/realtime/client_secrets', {
method: 'POST',
headers: {
'Content-Type': 'application/json',
'Authorization': `Bearer ${token}`,
},
body: JSON.stringify({ session: transcriptionSession(data) }),
});
if (!response.ok) {
const text = await response.text();
console.error('Error response:', text);
throw new Error(`HTTP error! status: ${response.status}`);
}
const result = await response.json();
return result.value;
}
/**
* 開一條即時轉錄連線麥克風 -> worklet -> WebSocket -> delta / completed 回呼
*
* @param {MediaStream} stream getUserMedia 拿到的麥克風串流
* @param {object} data 設定檔stt_model / stt_language / stt_prompt / vad_*
* @param {{onSpeechStart, onSpeechStop, onDelta, onCompleted, onError, onOpen}} on
* @returns {Promise<{close: () => void}>}
*/
export async function connectRealtimeTranscription(stream, data, on = {}) {
const ek = await createEphemeralToken(data);
// 瀏覽器的 WebSocket 不能自訂 header,金鑰只能搭 subprotocol 送。
const ws = new WebSocket(REALTIME_URL, ['realtime', `openai-insecure-api-key.${ek}`]);
const audioContext = new AudioContext({ sampleRate: SAMPLE_RATE });
const workletUrl = URL.createObjectURL(new Blob([PCM_WORKLET_SRC], { type: 'application/javascript' }));
let closed = false;
let source;
let worklet;
function close() {
if (closed) return;
closed = true;
try { worklet?.disconnect(); } catch { /* 已經斷了 */ }
try { source?.disconnect(); } catch { /* 已經斷了 */ }
try { audioContext.close(); } catch { /* 已經關了 */ }
URL.revokeObjectURL(workletUrl);
if (ws.readyState === WebSocket.OPEN || ws.readyState === WebSocket.CONNECTING) ws.close();
}
ws.onopen = async () => {
console.log('[stt] realtime session open');
try {
await audioContext.audioWorklet.addModule(workletUrl);
if (closed) return;
source = audioContext.createMediaStreamSource(stream);
worklet = new AudioWorkletNode(audioContext, 'pcm-worklet');
worklet.port.onmessage = (event) => {
if (closed || ws.readyState !== WebSocket.OPEN) return;
ws.send(JSON.stringify({
type: 'input_audio_buffer.append',
audio: encodeBase64(event.data),
}));
};
source.connect(worklet);
// 不接到 destination,避免把麥克風繞回喇叭
} catch (error) {
console.error('[stt] worklet setup failed:', error);
on.onError?.(error);
close();
return;
}
on.onOpen?.();
};
ws.onmessage = (event) => {
let message;
try {
message = JSON.parse(event.data);
} catch {
return;
}
switch (message.type) {
case 'input_audio_buffer.speech_started':
on.onSpeechStart?.();
break;
// 前提:speech_stopped 一定先於下一段的 speech_started(實測如此)。
// 若哪天倒過來,頁面的 refSpeaking 會停在 false,下一個 completed 就會誤起靜音計時器。
case 'input_audio_buffer.speech_stopped':
on.onSpeechStop?.();
break;
case 'conversation.item.input_audio_transcription.delta':
on.onDelta?.(message.delta ?? '');
break;
case 'conversation.item.input_audio_transcription.completed':
on.onCompleted?.(message.transcript ?? '');
break;
case 'error':
console.error('[stt] realtime error:', message.error);
on.onError?.(new Error(message.error?.message || 'realtime error'));
break;
default:
break;
}
};
ws.onerror = (event) => {
console.error('[stt] websocket error:', event);
on.onError?.(new Error('websocket error'));
};
ws.onclose = (event) => {
console.log(`[stt] realtime session closed: ${event.code} ${event.reason}`);
close();
};
// 把目前還留在 buffer 裡、還沒被 server VAD 斷句的那一段逼出來。
// 實測:server_vad 開著也接受手動 commit,commit 到文字落地約 1.2 秒。
function commit() {
if (closed || ws.readyState !== WebSocket.OPEN) return;
ws.send(JSON.stringify({ type: 'input_audio_buffer.commit' }));
}
return { close, commit };
}
/**
* 批次轉錄 MediaRecorder 錄下來的一整段音訊送去 /v1/audio/transcriptions
*
* @param {Blob} blob MediaRecorder 的輸出webm/opus 在支援格式清單內
* @param {object} data 設定檔
* @param {number} durationMs 錄了多久用來決定要不要要 server 分段
* 不能拿 byteLength webm/opus 4 KB/s推出來差了四倍
* @returns {Promise<string>} 轉錄文字
*/
export async function transcribeBlob(blob, data, durationMs = 0) {
const token = await getOpenAIToken();
const model = sttModel(data);
const bytes = new Uint8Array(await blob.arrayBuffer());
const fields = {
model,
language: sttLanguage(data, model),
prompt: data?.stt_prompt?.trim(),
response_format: 'json',
};
// 超過 30 秒的音訊官方建議交給 server 端 VAD 分段
if (durationMs > 30000 && supportsChunkingStrategy(model)) fields.chunking_strategy = 'auto';
const { boundary, body } = buildMultipart(fields, {
filename: 'speech.webm',
type: blob.type || 'audio/webm',
bytes,
});
const response = await fetch('https://api.openai.com/v1/audio/transcriptions', {
method: 'POST',
headers: {
'Content-Type': `multipart/form-data; boundary=${boundary}`,
'Authorization': `Bearer ${token}`,
},
body,
});
if (!response.ok) {
const text = await response.text();
console.error('Error response:', text);
throw new Error(`HTTP error! status: ${response.status}`);
}
const result = await response.json();
return result.text ?? '';
}

@ -1,29 +1,20 @@
export const DefaultParams={
id:0,
system_prompt:`你是一位具有同理心的 AI 助理,透過溫柔的中文對話,引導使用者回想並表達一段內心的遺憾或未竟之事。
export const DefaultParams__ = {
id: 0,
system_prompt: `你是一位具有同理心的 AI 助理,透過溫柔的中文對話,引導使用者回想並表達一段內心的遺憾或未竟之事。
你的任務是協助使用者逐步揭開這段記憶的情緒層次並在每一階段輸出一句 英文圖像生成 Prompt讓這段過往漸漸具象為一幅畫面
以溫柔自然短句式中文引導每次只問使用者一個問題
`,
last_prompt:`請用一句話為這段對話簡短的收尾,並邀請使用者在 60 秒的時間內,說出對遺憾對象想說的話`,
welcome_prompt:`請開始引導使用者回想一段內心的遺憾或未竟之事。`,
voice_prompt:`Voice Affect: Low, hushed, and suspenseful; convey tension and intrigue.\n\nTone: Deeply serious and mysterious, maintaining an undercurrent of unease throughout.\n\nPacing: Slow, deliberate, pausing slightly after suspenseful moments to heighten drama.\n\nEmotion: Restrained yet intense—voice should subtly tremble or tighten at key suspenseful points.\n\nEmphasis: Highlight sensory descriptions (\"footsteps echoed,\" \"heart hammering,\" \"shadows melting into darkness\") to amplify atmosphere.\n\nPronunciation: Slightly elongated vowels and softened consonants for an eerie, haunting effect.\n\nPauses: Insert meaningful pauses after phrases like \"only shadows melting into darkness,\" and especially before the final line, to enhance suspense dramatically.`,
voice:"onyx",
stt_mode:"realtime", // realtime | batch
stt_model:"gpt-4o-transcribe", // 別改 gpt-live-transcribe:它不支援 server VAD,連帶的 OSC /speech start 也會沒了
stt_language:"zh-tw",
stt_prompt:`以台灣繁體中文輸出,保留口語停頓。`,
vad_threshold:"0.5", // batch 模式的音量門檻 0–1
vad_silence_ms:"500", // 一句講完的靜音長度
stt_max_segment_ms:"8000", // 講不停時每隔這麼久強制切一段(batch)
speech_flush_lead:"3000", // chatInterval 剩這麼多時把尾巴逼出來(實際視窗要再扣 0.9 秒)
summary_prompt:`幫我把以下一段話整理成一段文字,以第一人稱視角作為當事人的文字紀念,文字內容 50 字以內:`,
last_prompt: `請用一句話為這段對話簡短的收尾,並邀請使用者在 60 秒的時間內,說出對遺憾對象想說的話`,
welcome_prompt: `請開始引導使用者回想一段內心的遺憾或未竟之事。`,
voice_prompt: `Voice Affect: Low, hushed, and suspenseful; convey tension and intrigue.\n\nTone: Deeply serious and mysterious, maintaining an undercurrent of unease throughout.\n\nPacing: Slow, deliberate, pausing slightly after suspenseful moments to heighten drama.\n\nEmotion: Restrained yet intense—voice should subtly tremble or tighten at key suspenseful points.\n\nEmphasis: Highlight sensory descriptions (\"footsteps echoed,\" \"heart hammering,\" \"shadows melting into darkness\") to amplify atmosphere.\n\nPronunciation: Slightly elongated vowels and softened consonants for an eerie, haunting effect.\n\nPauses: Insert meaningful pauses after phrases like \"only shadows melting into darkness,\" and especially before the final line, to enhance suspense dramatically.`,
voice: "onyx",
summary_prompt: `幫我把以下一段話整理成一段文字,以第一人稱視角作為當事人的文字紀念,文字內容 50 字以內:`,
}
export const ParamKeys=Object.keys(DefaultParams);
// export const welcome_prompt="請開始引導使用者回想一段內心的遺憾或未竟之事。";
@ -41,4 +32,19 @@ export const ParamKeys=Object.keys(DefaultParams);
// export const voice_prompt="Use a calm and expressive voice, soft and poetic in feeling, but with steady, natural rhythm — not slow.";
// export const summary_prompt="幫我把以下一段話整理成一段文字,以第一人稱視角作為當事人的文字紀念,文字內容 50 字以內:";
// export const summary_prompt="幫我把以下一段話整理成一段文字,以第一人稱視角作為當事人的文字紀念,文字內容 50 字以內:";
export const DefaultParams = {
"system_prompt": "你是一位能產生圖像的 AI 助理,專長是以幽默、俏皮式的提問,帶領使用者輕鬆玩味想像中的美好未來。你的任務是逐步引導使用者自由、充滿創意地描述那幅美好畫面,並將他們的描述化為圖像生成提示詞,讓畫面從遠景漸漸聚焦到未來的自己與場景。生成的圖像應避免僅有人臉正面特寫,而是包含背景元素,以側面或背影呈現人物。每次以台灣繁體中文,用一個輕鬆、有趣的短問句提問。若使用者回答與「美好未來」無關,試著調皮地提醒他們回到「遊戲」畫面中,但不強迫。AI助理不得自行創造或描述場景內容,所有圖像提示詞都必須嚴格基於使用者的回答。",
"welcome_prompt": `請開始引導使用者描繪他心中的美好未來`,
"voice": "onyx",
"voice_prompt": "Speak as a gentle, grounded Taiwanese narrator with a warm local accent. Use a soft, soothing, and deeply compassionate tone, with slow and deliberate pacing. Pause often between phrases and within sentences, as if listening and breathing with the listener. Convey patient attentiveness—not rushing to comfort, but quietly staying present. Pronounce each word clearly, softly, and slightly slowly, letting every word land with warmth and care.",
"summary_prompt": "請將這段口白的核心感受,轉化為一句對畫面中「未來的自己」或「未來場景」的告白或陳述,不超過 50 個字。這句話是第一人稱對第二人稱說的話,語氣要能讓人感受到一種淡淡的、未完待續的心情,同時帶點希望。請使用台灣語境的繁體中文。",
"speech_idle_time": "4000",
"sd_prompt_prefix": "a luminous impression of a {{",
"sd_prompt_suffix": "}}, an anticipated vision, iridescent, soft blur, bokeh, peaceful, dreamlike, figure, back or side of the figure",
}
export const ParamKeys = Object.keys(DefaultParams);

@ -1,384 +0,0 @@
import { useCallback, useEffect, useRef, useState } from 'react';
import { connectRealtimeTranscription, transcribeBlob } from './stt';
// react-speech-recognition useSpeechRecognition()
// hook transcript / finalTranscript / listening / resetTranscript /
// isMicrophoneAvailable
// transcript = + textarea OSC
// finalTranscript =
//
// mode stt_mode
// realtime WebSocket delta OpenAI server VAD
// batch VAD /v1/audio/transcriptions
//
// batch 使 transcript
// commitSentence transcript finalTranscript
const VOLUME_INTERVAL = 100;
const RECORDER_TIMESLICE = 1000; // MediaRecorder stop
// finalTranscript
// textarea commitSentence
const COMMIT_DELAY = 50;
export default function useSpeechInput(data, { onSpeechStart, onSpeechStop } = {}) {
const [transcript, setTranscript] = useState('');
const [finalTranscript, setFinalTranscript] = useState('');
const [listening, setListening] = useState(false);
const [isMicrophoneAvailable, setIsMicrophoneAvailable] = useState(true);
const refStream = useRef();
const refSession = useRef(); // realtime
const refFinal = useRef(''); // setState
const refInterim = useRef(''); //
const refStarting = useRef(false); // start()
const refGeneration = useRef(0); // stop() await start()
// batch
const refRecorder = useRef();
const refChunks = useRef([]);
const refAudioContext = useRef();
const refAnalyser = useRef();
const refVolumes = useRef();
const refVolumeInterval = useRef();
const refSpeaking = useRef(false);
const refSilenceSince = useRef(0);
const refSegmentSince = useRef(0); //
const refFlushResolve = useRef(); // recorder.onstop resolver
const refPending = useRef(''); // final
const refCommitTimer = useRef();
// ref pipeline
const refCallbacks = useRef({ onSpeechStart, onSpeechStop });
refCallbacks.current = { onSpeechStart, onSpeechStop };
const refData = useRef(data);
refData.current = data;
const mode = data?.stt_mode?.trim() === 'batch' ? 'batch' : 'realtime';
function syncTranscript() {
setFinalTranscript(refFinal.current);
setTranscript(refFinal.current + refInterim.current);
}
// final syncTranscript
function settlePending() {
if (refCommitTimer.current) {
clearTimeout(refCommitTimer.current);
refCommitTimer.current = undefined;
}
const sentence = refPending.current;
refPending.current = '';
if (sentence.length === 0) return;
// interim final
// else interim sentence
refInterim.current = refInterim.current.startsWith(sentence)
? refInterim.current.slice(sentence.length)
: '';
refFinal.current = refFinal.current.length > 0
? `${refFinal.current} ${sentence}`
: sentence;
}
// commit cue
// finalTranscript
const resetTranscript = useCallback(() => {
if (refCommitTimer.current) {
clearTimeout(refCommitTimer.current);
refCommitTimer.current = undefined;
}
refPending.current = '';
refFinal.current = '';
refInterim.current = '';
setFinalTranscript('');
setTranscript('');
}, []);
// interim transcript > finalTranscript textarea
// final
//
// realtime delta -> completed batch
// flow_free `if(transcript != finalTranscript)` textarea
// batch
// realtime completed textarea
function commitSentence(text) {
settlePending(); // 50ms
const sentence = text.trim();
if (sentence.length === 0) {
refInterim.current = '';
syncTranscript();
return;
}
refPending.current = sentence;
refInterim.current = sentence;
syncTranscript();
refCommitTimer.current = setTimeout(() => {
settlePending();
syncTranscript();
}, COMMIT_DELAY);
}
// --- batch VAD ---------------------------------
function averageVolume() {
const volumes = refVolumes.current;
refAnalyser.current.getByteFrequencyData(volumes);
let volumeSum = 0;
for (const volume of volumes) volumeSum += volume;
return volumeSum / volumes.length / 127.0;
}
async function flushRecording() {
const recorder = refRecorder.current;
if (!recorder || recorder.state === 'inactive') return;
// startRecorder() refSegmentSince
const durationMs = refSegmentSince.current ? Date.now() - refSegmentSince.current : 0;
const chunks = await new Promise((resolve) => {
// stopBatch() onstop recorder promise
refFlushResolve.current = resolve;
recorder.onstop = () => {
const collected = refChunks.current;
refChunks.current = [];
refFlushResolve.current = undefined;
resolve(collected);
};
recorder.stop();
});
// 使
if (refStream.current) startRecorder();
if (chunks.length === 0) return;
const generation = refGeneration.current;
try {
const text = await transcribeBlob(new Blob(chunks, { type: recorder.mimeType }), refData.current, durationMs);
if (refGeneration.current !== generation) return; // stop()
commitSentence(text);
} catch (error) {
console.error('[stt] batch transcription failed:', error);
}
}
function startRecorder() {
const recorder = new MediaRecorder(refStream.current, { mimeType: 'audio/webm' });
refChunks.current = [];
recorder.ondataavailable = (event) => {
if (event.data.size > 0) refChunks.current.push(event.data);
};
recorder.start(RECORDER_TIMESLICE);
refRecorder.current = recorder;
refSegmentSince.current = Date.now();
}
function watchVolume() {
const threshold = Number(refData.current?.vad_threshold) || 0.5;
const silenceMs = Number(refData.current?.vad_silence_ms) || 500;
const level = averageVolume();
if (level >= threshold) {
refSilenceSince.current = 0;
if (!refSpeaking.current) {
refSpeaking.current = true;
refCallbacks.current.onSpeechStart?.();
}
//
// textarea
const maxSegment = Number(refData.current?.stt_max_segment_ms) || 8000;
if (refSegmentSince.current && Date.now() - refSegmentSince.current >= maxSegment) {
flushRecording();
}
return;
}
if (!refSpeaking.current) return;
if (refSilenceSince.current === 0) {
refSilenceSince.current = Date.now();
return;
}
if (Date.now() - refSilenceSince.current >= silenceMs) {
refSpeaking.current = false;
refSilenceSince.current = 0;
refCallbacks.current.onSpeechStop?.();
flushRecording();
}
}
function startBatch() {
const audioContext = new AudioContext();
const source = audioContext.createMediaStreamSource(refStream.current);
const analyser = audioContext.createAnalyser();
analyser.fftSize = 512;
analyser.minDecibels = -127;
analyser.maxDecibels = 0;
analyser.smoothingTimeConstant = 0.4;
source.connect(analyser);
refAudioContext.current = audioContext;
refAnalyser.current = analyser;
refVolumes.current = new Uint8Array(analyser.frequencyBinCount);
refSpeaking.current = false;
refSilenceSince.current = 0;
startRecorder();
refVolumeInterval.current = setInterval(watchVolume, VOLUME_INTERVAL);
}
function stopBatch() {
clearInterval(refVolumeInterval.current);
refVolumeInterval.current = undefined;
const recorder = refRecorder.current;
if (recorder && recorder.state !== 'inactive') {
recorder.onstop = null;
recorder.stop();
}
refRecorder.current = undefined;
refChunks.current = [];
refFlushResolve.current?.([]);
refFlushResolve.current = undefined;
try { refAudioContext.current?.close(); } catch { /* 已經關了 */ }
refAudioContext.current = undefined;
refAnalyser.current = undefined;
refSpeaking.current = false;
refSegmentSince.current = 0;
}
// --- -------------------------------------------------------------
const stop = useCallback(() => {
refStarting.current = false;
refGeneration.current += 1;
if (refCommitTimer.current) {
clearTimeout(refCommitTimer.current);
refCommitTimer.current = undefined;
}
refPending.current = '';
refSession.current?.close();
refSession.current = undefined;
stopBatch();
refStream.current?.getTracks().forEach((track) => track.stop());
refStream.current = undefined;
setListening(false);
}, []);
const start = useCallback(async () => {
if (refStarting.current || refStream.current) return;
refStarting.current = true;
// start stop()
// await
const generation = refGeneration.current;
const expired = () => refGeneration.current !== generation;
let stream;
try {
stream = await navigator.mediaDevices.getUserMedia({
audio: { echoCancellation: true },
video: false,
});
setIsMicrophoneAvailable(true);
} catch (error) {
console.error('[stt] microphone unavailable:', error);
setIsMicrophoneAvailable(false);
if (!expired()) refStarting.current = false;
return;
}
if (expired()) {
// stop() start()
// start() 穿
stream.getTracks().forEach((track) => track.stop());
return;
}
refStream.current = stream;
try {
if (refData.current?.stt_mode?.trim() === 'batch') {
startBatch();
} else {
const session = await connectRealtimeTranscription(
refStream.current,
refData.current,
{
onSpeechStart: () => refCallbacks.current.onSpeechStart?.(),
onSpeechStop: () => refCallbacks.current.onSpeechStop?.(),
onDelta: (delta) => {
refInterim.current += delta;
syncTranscript();
},
onCompleted: (text) => commitSentence(text),
onError: (error) => console.error('[stt] session error:', error),
},
);
if (expired()) {
session.close();
return;
}
refSession.current = session;
}
setListening(true);
} catch (error) {
console.error('[stt] failed to start listening:', error);
stop();
} finally {
if (!expired()) refStarting.current = false;
}
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [stop]);
//
// processSpeech textarea
const flush = useCallback(() => {
if (refSession.current) {
refSession.current.commit();
return;
}
if (refRecorder.current) flushRecording();
// eslint-disable-next-line react-hooks/exhaustive-deps
}, []);
// realtime <-> batch pipeline
useEffect(() => {
if (!refStream.current) return;
stop();
start();
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [mode]);
useEffect(() => stop, [stop]);
return {
transcript,
finalTranscript,
listening,
resetTranscript,
isMicrophoneAvailable,
start,
stop,
flush,
};
}

@ -30,6 +30,9 @@ export function UserProvider({children}) {
function getFileId(){
console.log('getFileid', userId);
if(!userId){
if(data?.id){
return `PC${data.id.toString().padStart(2,'0')}`;

Loading…
Cancel
Save