diff --git a/.gitignore b/.gitignore index 0cbd177..1ecff26 100644 --- a/.gitignore +++ b/.gitignore @@ -80,3 +80,10 @@ tmp/ *.tmp.mov *.tmp.avi *.scratch.* + +runtime.local.json +devices.json +certs/ + +# 漏斗的语音提示音是运行时必需资源,必须随仓库分发 +!extensions/assistive_harness/phase_b/assets/reject_wav/*.wav diff --git a/README.md b/README.md index df9ce6e..2bf698a 100644 --- a/README.md +++ b/README.md @@ -120,7 +120,7 @@ cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=ON -DLLAMA_CURL=OFF cmake --build build --config Release --target llama-omni-server -j cd .. -git clone --branch master https://github.com/OpenBMB/MiniCPM-o-Demo.git +git clone --branch main https://github.com/OpenBMB/MiniCPM-o-Demo.git cd MiniCPM-o-Demo python -m pip install -r requirements.txt cd .. @@ -128,6 +128,9 @@ cd .. git clone https://github.com/OpenSQZ/OpenGlass.git cd OpenGlass python -m pip install -r runtime/openglass_omni/requirements.txt + +python -m pip install -r extensions/assistive_harness/phase_b/requirements-phase-b.txt # voice control + CV funnel (harness chains only; skip if you just run basic chat) + ``` Because upstream `master` branches can change, record the exact commit SHAs used for every validated OpenSQZ Glass runtime release. Upstream changes may alter ports, arguments, protocols, TTS behavior, or process ownership. diff --git a/README_zh.md b/README_zh.md index 6f5f855..ba13d70 100644 --- a/README_zh.md +++ b/README_zh.md @@ -92,7 +92,7 @@ flowchart LR 鎺у埗闈㈡澘鐨勭洰鏍囨槸锛氬畬鎴愪竴娆℃х幆澧冨噯澶囧悗锛屽悗缁疄楠屽彲浠ヤ竴閿噸澶嶅惎鍔ㄣ傚畠涓嶄細鏇跨敤鎴蜂笅杞芥ā鍨嬫潈閲嶃乧lone 涓婃父浠撳簱銆佺紪璇 `llama.cpp-omni` 鎴栫儳褰 ESP32銆 -> **褰撳墠鍏ㄦ柊 clone 鐘舵侊細** 鍙互浠庝粨搴撴牴鐩綍鍚姩闈㈡澘 UI锛屼絾褰撳墠鍚姩鍣ㄤ粛浠 `runtime/openglass_omni/panel.py` 璇诲彇鏈哄櫒涓撳睘璺緞銆備粨搴撳凡鏈夌殑 `runtime.local.json` 鍔犺浇鍣ㄥ皻鏈帴鍏ヨ闈㈡澘銆傝鎸夌収涓嬮潰鍒楀嚭鐨勫疄闄呯敓鏁堜綅缃厤缃紱鍦ㄥ畬鎴愪笅涓杞唬鐮佷慨鏀瑰墠锛岃繕涓嶈兘鎶婂畠绉颁负鍙Щ妞嶇殑涓閿畨瑁呫 +> **褰撳墠鍏ㄦ柊 clone 鐘舵侊細** 浠庝粨搴撴牴鐩綍鍚姩闈㈡澘 UI锛屾湰鏈鸿矾寰勫啓鍦 `runtime.local.json`锛堝鍒 `runtime.example.json` 寰楀埌锛夛紝闈㈡澘鍚姩鏃惰鍙栥傜溂闀滃垪琛ㄥ拰鏃嬭浆瑙掑湪 `devices.json`銆俆LS 璇佷功棣栨鍚姩鑷姩鐢熸垚銆備粛闇鎵嬪姩鍑嗗鐨勬槸浠撳簱澶栫殑璧勬簮锛歚llama.cpp-omni` 缂栬瘧缁撴灉銆丮iniCPM-o GGUF 鏉冮噸銆丗unASR 妯″瀷锛屼互鍙 ESP32 鍥轰欢閲岀殑 Wi-Fi 鍑嵁銆 ### 1. 鍓嶇疆鏉′欢 @@ -107,11 +107,11 @@ flowchart LR 鏈粨搴撲笉鍒嗗彂妯″瀷鏉冮噸銆 -### 2. Clone V2 涓婃父椤圭洰鐨 master 鍒嗘敮 +### 2. Clone 涓婃父椤圭洰鐨 master 鍒嗘敮 涓変釜浠撳簱蹇呴』淇濇寔鐩镐簰鐙珛锛屼笉瑕佹妸 OpenSQZ Glass 鐨勬枃浠跺鍒惰繘 MiniCPM-o-Demo銆 -褰撳墠鍏紑鍚姩鍣ㄩ噰鐢 V2 鍥涜繘绋嬮摼璺細`llama-omni-server` -> `worker` -> `gateway` -> `demo`锛屽搴斾袱涓笂娓搁」鐩粛鍦ㄧ淮鎶ょ殑 `master` 鍒嗘敮銆俉AIC 婕旂ず浣跨敤鐨 V1 涓夎繘绋嬮摼璺敱 `worker` 鑷鍚姩 `llama-server`锛屽睘浜庡巻鍙茶繍琛屾柟妗堬紝涓嶅啀浣滀负鏈 README 鐨勯粯璁ゅ畨瑁呰矾寰勩 +褰撳墠鍏紑鍚姩鍣ㄩ噰鐢ㄤ笌闈㈠鍚屾鐨勫洓杩涚▼閾捐矾锛歚llama-omni-server` -> `worker` -> `gateway` -> `demo`锛屽搴斾袱涓笂娓搁」鐩粛鍦ㄧ淮鎶ょ殑 `master` 鍒嗘敮銆 ```powershell git clone --branch master https://github.com/tc-mb/llama.cpp-omni.git @@ -120,7 +120,7 @@ cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=ON -DLLAMA_CURL=OFF cmake --build build --config Release --target llama-omni-server -j cd .. -git clone --branch master https://github.com/OpenBMB/MiniCPM-o-Demo.git +git clone --branch main https://github.com/OpenBMB/MiniCPM-o-Demo.git cd MiniCPM-o-Demo python -m pip install -r requirements.txt cd .. @@ -128,8 +128,13 @@ cd .. git clone https://github.com/OpenSQZ/OpenGlass.git cd OpenGlass python -m pip install -r runtime/openglass_omni/requirements.txt + +python -m pip install -r extensions/assistive_harness/phase_b/requirements-phase-b.txt # 璇煶鎺у埗 + CV 婕忔枟锛坔arness閾捐矾闇瑕侊紝鍙窇鍩虹瀵硅瘽鏃犻渶瀹夎锛 + ``` + + 鐢变簬涓婃父 `master` 鍒嗘敮浼氭寔缁彉鍖栵紝姣忔瀹屾垚 OpenSQZ Glass 杩愯鏃堕獙璇佸拰姝e紡鍙戝竷鏃讹紝閮藉簲璁板綍瀹為檯娴嬭瘯杩囩殑 Commit SHA銆備笂娓告洿鏂板彲鑳芥敼鍙樼鍙c佸惎鍔ㄥ弬鏁般侀氫俊鍗忚銆乀TS 琛屼负鎴栬繘绋嬫墍鏈夋潈銆 ### 3. 鍑嗗妯″瀷鏂囦欢 @@ -177,17 +182,33 @@ Copy-Item examples/configs/devices.example.json runtime/openglass_omni/devices.j ### 5. 閰嶇疆褰撳墠鍚姩鍣 -褰撳墠鐗堟湰鐪熸鐢熸晥鐨勬槸浠ヤ笅閰嶇疆锛 +鎶 [`runtime.example.json`](runtime/openglass_omni/runtime.example.json) 澶嶅埗涓 +鍚岀洰褰曚笅鐨 `runtime.local.json`锛屽~鍐欐湰鏈鸿矾寰勩傞潰鏉垮惎鍔ㄦ椂璇诲彇瀹冿紝 +鐢ㄥ叾涓殑鍊艰鐩 `panel.py` 閲岀殑榛樿璺緞锛 -| 閰嶇疆鍐呭 | 褰撳墠瀹為檯鐢熸晥浣嶇疆 | 搴斿~鍐欑殑鍊 | +```json +{ + "conda_env": null, + "minicpm_demo_root": "D:\\path\\to\\MiniCPM-o-Demo", + "llama_server": "D:\\path\\to\\llama.cpp-omni\\build\\bin\\Release\\llama-omni-server.exe", + "llama_model": "D:\\path\\to\\MiniCPM-o-gguf\\MiniCPM-o-4_5-Q4_K_M.gguf", + "asr_model": "D:\\path\\to\\LocalASRmodel\\speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-online", + "glasses": { "ssid": "浣犵殑WiFi", "psk": "浣犵殑WiFi瀵嗙爜" } +} +``` + +| 閰嶇疆鍐呭 | 浣嶇疆 | 璇存槑 | | --- | --- | --- | -| MiniCPM-o-Demo 鐩綍 | [`panel.py` 鐨 `CONFIG["minicpm_demo_dir"]`](runtime/openglass_omni/panel.py) | 鍖呭惈涓婃父 `worker.py` 鍜 `gateway.py` 鐨勭粷瀵硅矾寰 | -| `llama-omni-server` | [`panel.py` 鐨 `CONFIG["procs"]["llama"]`](runtime/openglass_omni/panel.py) | `llama.cpp-omni/build` 涓嬬殑缂栬瘧缁撴灉 | -| 涓 GGUF 妯″瀷 | 鍚屼竴鏉 `llama` 鍛戒护涓 `-m` 鍚庣殑浣嶇疆 | MiniCPM-o 4.5 涓 GGUF 鐨勭粷瀵硅矾寰 | +| MiniCPM-o-Demo 鐩綍 | `runtime.local.json` 鐨 `minicpm_demo_root` | 鍖呭惈涓婃父 `worker.py` / `gateway.py` 鐨勭粷瀵硅矾寰 | +| `llama-omni-server` | `llama_server` | `llama.cpp-omni/build` 涓嬬殑缂栬瘧缁撴灉 | +| 涓 GGUF 妯″瀷 | `llama_model` | MiniCPM-o 4.5 涓 GGUF 鐨勭粷瀵硅矾寰 | +| ASR 妯″瀷 | `asr_model` | FunASR 娴佸紡妯″瀷鐩綍锛屸憽鈶⑩懀 閾捐矾鐨勮闊虫帶鍒剁敤 | +| 鐪奸暅 Wi-Fi | `glasses.ssid` / `glasses.psk` | Rokid 閾捐矾鍚姩 APK 鏃朵紶鍏ワ紱鐣欑┖鍒欒閾捐矾杩炰笉涓婄溂闀滐紝鍏朵綑閾捐矾涓嶅彈褰卞搷 | | 鐪奸暅鍚嶇О/IP/鏃嬭浆瑙 | `runtime/openglass_omni/devices.json` | 姣忓壇 ESP32 鐪奸暅涓鏉¤褰 | -| Prompt 棰勮 | [`panel.py` 鐨 `CONFIG["presets"]`](runtime/openglass_omni/panel.py) | 褰撳墠闈㈡澘涓樉绀虹殑浜や簰 Prompt | +| Prompt 棰勮 | [`panel.py` 鐨 `CONFIG["presets"]`](runtime/openglass_omni/panel.py) | 闈㈡澘涓樉绀虹殑浜や簰 Prompt | -[`runtime.example.json`](runtime/openglass_omni/runtime.example.json) 鍜 [`prompts.json`](runtime/openglass_omni/prompts.json) 鎻忚堪浜嗘垜浠噯澶囬噰鐢ㄧ殑鏈湴閰嶇疆杈圭晫锛屼絾褰撳墠闈㈡澘杩樻病鏈夎鍙栬繖涓や釜鏂囦欢銆傛妸 runtime 绀轰緥澶嶅埗涓 `runtime.local.json` **杩樹笉鑳芥浛浠** `panel.py` 涓啓姝荤殑璺緞鍜 Prompt銆傝繖鏄凡鐭ョ殑闆嗘垚闂锛屼笉鏄敤鎴烽厤缃敊璇 +**TLS 璇佷功涓嶇敤鎵嬪姩鍑嗗銆** 闈㈡澘棣栨鍚姩鏃朵細鍦 `/certs/` 涓嬭嚜绛句竴瀵 +锛坓ateway 8006 鍜 harness 8021 鍏辩敤锛夛紝宸插瓨鍦ㄥ垯涓嶈鐩栥 ### 6. 閰嶇疆骞剁儳褰 Wi-Fi 鍥轰欢 @@ -203,16 +224,35 @@ Copy-Item examples/configs/devices.example.json runtime/openglass_omni/devices.j python glasses_panel.py ``` -閫夋嫨 **ESP32 鐪奸暅**锛屽啀閫夋嫨璁惧鍚嶇О骞剁偣鍑 **涓閿惎鍔**銆傚綋鍓嶉潰鏉夸細灏濊瘯渚濇鍚姩锛 +鍦ㄩ《閮ㄩ夋嫨**閾捐矾**銆**鐪奸暅**锛堚憿鈶 杩樿閫**鍒ゆ嵁**锛夛紝鐐瑰嚮 **涓閿惎鍔**銆 + +闈㈡澘鎻愪緵浜旀潯閾捐矾锛屽叾涓 ESP32 鐨勫洓妗f槸閫掕繘鐨勶紝姣忎竴妗e彧姣斾笂涓妗e涓浠朵簨锛 + +| 閾捐矾 | 鍔熻兘 | 鍚姩鐨勮繘绋 | +| --- | --- | --- | +| 鈶 鍩虹瀵硅瘽 | 閫氳繃鐪奸暅涓庢湰鍦版ā鍨嬪叏妯℃佸弻宸ュ璇 | llama 鈫 worker 鈫 gateway 鈫 esp32_bridge | +| 鈶 璇煶鎺у埗 | 鑳藉惉鎳傘屽仠涓涓 / 閲嶆柊寮濮 / 鎵句笢瑗裤 | llama 鈫 worker 鈫 gateway 鈫 harness鈫抏sp32_bridge | +| 鈶 璐ㄩ噺绛涢 | 姣忕澶氬抚閲屾寫鏈娓呮櫚鐨勪竴寮犻佹ā鍨 | 鍚 鈶 | +| 鈶 瀹屾暣闃插够瑙 | 鍧忓浘鐩存帴鎷︿笅骞惰闊虫彁绀猴紝涓嶈妯″瀷鐪嬭 | 鍚 鈶 | +| Rokid | 鍙嶅悜閾捐矾锛孭C 寮 18080 绛 APK 杩炲叆 | llama 鈫 worker 鈫 gateway 鈫 rokid | + +鈶狅綖鈶 鏄**浜掓枼**鐨勶細瀹冧滑閮借鐙崰鐪奸暅鐨勯煶棰戦氶亾鍜屽浘鍍忕鍙o紝鍚屼竴鏃堕棿鍙兘璺戜竴鏉° ```text llama-omni-server :22500 -> worker :22400 -> gateway :8006 - -> ESP32 bridge / 鏈湴鐢婚潰 :8080 + -> 锛堚憽鈶⑩懀 鎵嶆湁锛塰arness :8021 + -> 鐪奸暅瀹㈡埛绔 / 绗竴瑙嗚 :8080 ``` -鍙湁鍥涗釜杩涚▼鎸囩ず鐏叏閮ㄥ彉缁匡紝骞朵笖绗竴瑙嗚鎸佺画鏇存柊锛屾墠鑳借鏄庨摼璺氨缁傚彧鐪嬪埌闈㈡澘 UI 鎵撳紑锛屼笉鑳借瘉鏄庢ā鍨嬨佸0闊炽佸浘鍍忓拰鍝嶅簲閾捐矾宸茬粡璺戦氥 +**鍒ゆ嵁妗d綅**锛堝彧鏈 鈶⑩懀 闇瑕侀夛級鍐冲畾婕忔枟鐢ㄥ摢濂楁爣瀹氬弬鏁帮細 + +- **涓ユ牸鍒ゆ嵁** 鈥斺 鏈夊畨鍏ㄩ闄╃殑鍦烘櫙锛屼緥濡傞渶瑕佸康瀛楃殑鑽洅銆佽矾涓婃寚绀虹墝绛夈傝鍒ゆ嵁涓ユ牸鎷掔粷璐ㄩ噺涓嶄匠鐨勫浘鍍忛槻姝㈡ā鍨嬭緭鍑哄够瑙夈 +- **鏃ュ父鍒ゆ嵁** 鈥斺 鏃ュ父鍦烘櫙銆 + +鍙湁杩涚▼鎸囩ず鐏叏閮ㄥ彉缁裤佸苟涓旂涓瑙嗚鎸佺画鏇存柊锛屾墠鑳借鏄庨摼璺氨缁 +鍙湅鍒伴潰鏉 UI 鎵撳紑锛屼笉鑳借瘉鏄庢ā鍨嬨佸0闊炽佸浘鍍忓拰鍝嶅簲閾捐矾宸茬粡璺戦氥 ### 闈㈡澘杩涚▼鐢熷懡鍛ㄦ湡 @@ -258,8 +298,8 @@ OpenGlass/ ## 宸茬煡闄愬埗 -- 褰撳墠鍏紑浠撳簱灏氭湭瀹屾垚鍏ㄦ柊鏈哄櫒涓婄殑 Omni 绔埌绔獙璇併 -- 褰撳墠闈㈡澘浠嶅寘鍚満鍣ㄤ笓灞炶繍琛岃矾寰勶紝娌℃湁鐪熸浣跨敤 `runtime.local.json`銆 +- 璇煶鎶鑳藉垏鎹㈡椂鐨勬寚浠ゆ敞鍏ュ湪褰撳墠 `/v1/realtime` 鍗忚涓嬩笉鐢熸晥锛堝崗璁病鏈夊搴斿瓧娈碉級锛屾ā鍨嬪彧鑳戒緷璧 system prompt 瀹屾垚浠诲姟銆 +- 銆屽熀纭瀵硅瘽銆嶄笌 harness 閾捐矾鍚勮嚜甯︿竴濂楃涓瑙嗚鍓嶇鍜屽綍鍒舵ā鍧楋紝浠撳簱涓瓨鍦ㄤ袱浠藉悓鍚嶆枃浠讹紝浜掍笉褰卞搷浣嗗鏄撴贩娣嗐 - Prompt 浠嶅啓鍦 `panel.py` 涓紝鐙珛鐨 `prompts.json` 灏氭湭鎺ュ叆銆 - ESP32 Wi-Fi 浠嶉渶淇敼 tracked `.ino`锛屾湰鍦 Wi-Fi 澶存枃浠舵ā鏉垮皻鏈帴鍏ャ - 姝e父鍏崇獥浼氭墽琛屾竻鐞嗭紝浣嗗紓甯搁鍑哄彲鑳界暀涓嬪瓙杩涚▼鎴栧閮ㄥ惎鍔ㄧ殑杩涚▼銆 diff --git a/extensions/assistive_harness/.gitignore b/extensions/assistive_harness/.gitignore new file mode 100644 index 0000000..e607f1c --- /dev/null +++ b/extensions/assistive_harness/.gitignore @@ -0,0 +1,4 @@ +__pycache__/ +*.pyc +runs/ +test_artifacts/ diff --git a/extensions/assistive_harness/OPENGLASS_MIGRATION_MAP.md b/extensions/assistive_harness/OPENGLASS_MIGRATION_MAP.md new file mode 100644 index 0000000..1110736 --- /dev/null +++ b/extensions/assistive_harness/OPENGLASS_MIGRATION_MAP.md @@ -0,0 +1,25 @@ +# OpenGlass Migration Map + +Phase A deliberately separates portable policy from browser-specific transport. + +| Phase A component | OpenGlass-side destination | +|---|---| +| `schemas.py` | shared structured control-event schema | +| `router.py` | local deterministic voice-command router | +| `registry.py` + prompts | configurable Skill registry/prompt bundle | +| `echo_guard.py` | output/playback-aware ASR echo filter | +| `state_machine.py` | device-independent control priority and dedupe | +| `asr/` | phone/host local ASR service behind the same interface | +| `cv/base.py` | future advisory perception provider interface | +| `server.py` WS messages | replaceable host/phone transport adapter | +| `browser-session-adapter.js` | reference semantics for an OpenGlass session adapter | + +OpenGlass migration should keep `ControlEvent`, `SkillRegistry`, prompt SHA, +generation fence and STOP priority stable. Replace only AudioMirror capture, +session lifecycle calls, and the control transport. No dependency on ESP32, Rokid, +DOM layout or the MiniCPM private wire format exists in the Python core. + +The first Rokid implementation of this boundary now lives in +[`phase_b/`](phase_b/README.md). It consumes the existing APK JPEG/PCM endpoints, +reuses this Core and the existing Gateway, and implements the browser adapter's +STOP/RESUME/RESET/Skill generation contract in Python. diff --git a/extensions/assistive_harness/README.md b/extensions/assistive_harness/README.md new file mode 100644 index 0000000..8be88a5 --- /dev/null +++ b/extensions/assistive_harness/README.md @@ -0,0 +1,60 @@ +# Assistive Voice Skill Harness (Phase A) + +This directory is an optional, local sidecar for the MiniCPM-o browser Demo. It is +disabled by default and does not replace the native microphone/video path. + +Start the sidecar explicitly: + +```powershell +python -m extensions.assistive_harness.server --enabled ` + --model-path "C:\path\to\a\local\FunASR\model" +``` + +Then opt the browser tab in with `?assistive_harness=1`. Test-only transcript +injection additionally requires `--allow-test-injection`; it is never enabled by +the normal command above. + +The browser hook is deliberately thin. Control, ASR, prompt registry, echo guard, +state machine, metrics, and CV shadow interfaces live under this directory so the +module can later be moved behind another transport (for example OpenGlass). + +The first Rokid/OpenGlass transport is implemented in +[`phase_b/`](phase_b/README.md). It preserves this Core and the existing 8040 +Gateway while replacing browser microphone, camera, playback, and Session calls +with a Python device adapter. + +## User-editable Skills + +System prompts live in `extensions/assistive_harness/prompts/`. Existing prompt +files are read on every activation, so users can replace a prompt and activate +the Skill again without restarting the sidecar. Skill IDs, enable flags and +voice phrases are configured in `config/skills.example.yaml`; changing that YAML +does require restarting the sidecar. + +Enabled by default: + +- `甯垜鎵<鐗╀綋>` -> `find_object` -> hot Session restart with `{{target}}` +- `璇讳竴涓媊 / `甯垜璇嗗瓧` -> `read_text` -> hot Session restart +- `鎻忚堪涓涓媊 / `鐪嬬湅鍛ㄥ洿` -> `describe_scene` -> hot Session restart +- `甯垜閬块殰` / `鍓嶉潰鏈夐殰纰嶅悧` -> `obstacle_avoidance` -> hot Session restart +- `鍥炲埌鑱婂ぉ` / `鎭㈠鏅氳亰澶ー -> `idle_chat` -> hot Session restart + +`obstacle_avoidance` contains the frozen AAAI_SI prompt and is enabled only for +stationary, supervised validation. Its mobility safety has not been accepted; +never treat this Demo as a navigation or safety device. + +With the sidecar running, `GET http://127.0.0.1:8021/skills` reports the active +configuration and absolute prompt path for every registered Skill. + +## CV V1 shadow pipeline + +Browser JPEG mirrors enter a capacity-one latest-frame queue and are analyzed +on a dedicated single-thread worker. `cv_mode: disabled` drops frames before +inference; `cv_mode: shadow` writes `CVObservation` records without changing a +Skill or taking ownership of MiniCPM output. Slow, timed-out, or failed plugins +remain outside the audio/control receive path. + +`find_object` uses the local `yolo_onnx` reference provider. Other Skills keep +the `noop` provider until a task-specific plugin is registered. Provider setup, +the observation schema, and an OCR implementation template are documented in +[`cv/README.md`](cv/README.md). diff --git a/extensions/assistive_harness/__init__.py b/extensions/assistive_harness/__init__.py new file mode 100644 index 0000000..339e7b9 --- /dev/null +++ b/extensions/assistive_harness/__init__.py @@ -0,0 +1,14 @@ +"""Portable voice-controlled Skill Harness core for Phase A.""" + +from .router import RuleIntentRouter +from .registry import SkillRegistry +from .schemas import ASREvent, ControlEvent, ControlIntent + +__all__ = [ + "ASREvent", + "ControlEvent", + "ControlIntent", + "RuleIntentRouter", + "SkillRegistry", +] + diff --git a/extensions/assistive_harness/asr/__init__.py b/extensions/assistive_harness/asr/__init__.py new file mode 100644 index 0000000..6d88083 --- /dev/null +++ b/extensions/assistive_harness/asr/__init__.py @@ -0,0 +1,6 @@ +from .base import ASREngine, ASRResult +from .energy_vad import EnergyVAD, UtteranceAudio +from .funasr_engine import FunASREngine + +__all__ = ["ASREngine", "ASRResult", "EnergyVAD", "FunASREngine", "UtteranceAudio"] + diff --git a/extensions/assistive_harness/asr/base.py b/extensions/assistive_harness/asr/base.py new file mode 100644 index 0000000..a6fc73d --- /dev/null +++ b/extensions/assistive_harness/asr/base.py @@ -0,0 +1,30 @@ +from __future__ import annotations + +from dataclasses import dataclass +from typing import Protocol + +import numpy as np + + +@dataclass(frozen=True, slots=True) +class ASRResult: + text: str + confidence: float + model: str + device: str + + +class ASREngine(Protocol): + def transcribe(self, audio: np.ndarray, sample_rate: int) -> ASRResult: + ... + + +class ScriptedASREngine: + def __init__(self, transcripts: list[str]): + self.transcripts = list(transcripts) + + def transcribe(self, audio: np.ndarray, sample_rate: int) -> ASRResult: + del audio, sample_rate + text = self.transcripts.pop(0) if self.transcripts else "" + return ASRResult(text=text, confidence=1.0, model="scripted", device="cpu") + diff --git a/extensions/assistive_harness/asr/energy_vad.py b/extensions/assistive_harness/asr/energy_vad.py new file mode 100644 index 0000000..58cec2e --- /dev/null +++ b/extensions/assistive_harness/asr/energy_vad.py @@ -0,0 +1,92 @@ +from __future__ import annotations + +from collections import deque +from dataclasses import dataclass + +import numpy as np + + +@dataclass(frozen=True, slots=True) +class UtteranceAudio: + audio: np.ndarray + started_at_ms: float + ended_at_ms: float + + +class EnergyVAD: + """Small streaming endpoint detector; it does not perform recognition.""" + + def __init__( + self, + sample_rate: int = 16_000, + rms_threshold: float = 0.012, + min_speech_ms: int = 180, + end_silence_ms: int = 450, + max_utterance_ms: int = 8_000, + preroll_ms: int = 200, + ): + self.sample_rate = sample_rate + self.rms_threshold = rms_threshold + self.min_speech_ms = min_speech_ms + self.end_silence_ms = end_silence_ms + self.max_utterance_ms = max_utterance_ms + self.preroll_ms = preroll_ms + self._preroll: deque[tuple[np.ndarray, float]] = deque() + self._active: list[np.ndarray] = [] + self._started_at_ms: float | None = None + self._last_voice_ms: float | None = None + + def _duration_ms(self, audio: np.ndarray) -> float: + return float(audio.size) * 1000.0 / self.sample_rate + + def feed(self, audio: np.ndarray, frame_started_at_ms: float) -> UtteranceAudio | None: + frame = np.asarray(audio, dtype=np.float32).reshape(-1).copy() + if frame.size == 0: + return None + duration_ms = self._duration_ms(frame) + frame_end_ms = frame_started_at_ms + duration_ms + rms = float(np.sqrt(np.mean(np.square(frame, dtype=np.float64)))) + voiced = rms >= self.rms_threshold + + if self._started_at_ms is None: + self._preroll.append((frame, frame_started_at_ms)) + while self._preroll and frame_end_ms - self._preroll[0][1] > self.preroll_ms: + self._preroll.popleft() + if not voiced: + return None + self._started_at_ms = self._preroll[0][1] if self._preroll else frame_started_at_ms + self._active = [item[0] for item in self._preroll] + self._preroll.clear() + self._last_voice_ms = frame_end_ms + else: + self._active.append(frame) + if voiced: + self._last_voice_ms = frame_end_ms + + active_ms = frame_end_ms - float(self._started_at_ms) + silence_ms = frame_end_ms - float(self._last_voice_ms or frame_end_ms) + if active_ms >= self.max_utterance_ms or silence_ms >= self.end_silence_ms: + return self._finish(frame_end_ms) + return None + + def _finish(self, ended_at_ms: float) -> UtteranceAudio | None: + if self._started_at_ms is None or not self._active: + self.reset() + return None + audio = np.concatenate(self._active).astype(np.float32, copy=False) + started = self._started_at_ms + voiced_duration = max(0.0, float(self._last_voice_ms or ended_at_ms) - started) + self.reset() + if voiced_duration < self.min_speech_ms: + return None + return UtteranceAudio(audio=audio, started_at_ms=started, ended_at_ms=ended_at_ms) + + def flush(self, ended_at_ms: float) -> UtteranceAudio | None: + return self._finish(ended_at_ms) + + def reset(self) -> None: + self._preroll.clear() + self._active = [] + self._started_at_ms = None + self._last_voice_ms = None + diff --git a/extensions/assistive_harness/asr/funasr_engine.py b/extensions/assistive_harness/asr/funasr_engine.py new file mode 100644 index 0000000..c7a126d --- /dev/null +++ b/extensions/assistive_harness/asr/funasr_engine.py @@ -0,0 +1,84 @@ +from __future__ import annotations + +import threading +from pathlib import Path +from typing import Any + +import numpy as np + +from .base import ASRResult + + +class FunASREngine: + """Lazy local FunASR adapter. It never downloads a model implicitly.""" + + def __init__(self, model_path: str, device: str = "cpu", model_kwargs: dict[str, Any] | None = None): + path = Path(model_path).expanduser().resolve() + if not path.is_dir(): + raise FileNotFoundError(f"FunASR model path does not exist: {path}") + self.model_path = str(path) + self.device = device + self.model_kwargs = dict(model_kwargs or {}) + self._model: Any = None + self._load_lock = threading.Lock() + self._infer_lock = threading.Lock() + + @property + def loaded(self) -> bool: + return self._model is not None + + def ensure_loaded(self) -> None: + if self._model is not None: + return + with self._load_lock: + if self._model is not None: + return + from funasr import AutoModel + + self._model = AutoModel( + model=self.model_path, + device=self.device, + disable_update=True, + **self.model_kwargs, + ) + + def warm_up(self) -> None: + """Load the model and run one short silent inference before serving clients.""" + self.ensure_loaded() + silence = np.zeros(8_000, dtype=np.float32) + with self._infer_lock: + self._model.generate( + input=silence, + cache={}, + is_final=True, + batch_size_s=0, + ) + + def transcribe(self, audio: np.ndarray, sample_rate: int) -> ASRResult: + if sample_rate != 16_000: + raise ValueError(f"FunASR Phase A requires 16 kHz audio, got {sample_rate}") + self.ensure_loaded() + waveform = np.asarray(audio, dtype=np.float32).reshape(-1) + with self._infer_lock: + results = self._model.generate( + input=waveform, + cache={}, + is_final=True, + batch_size_s=0, + ) + text = "" + confidence = 1.0 + if isinstance(results, list) and results: + first = results[0] + if isinstance(first, dict): + text = str(first.get("text") or "").strip() + if isinstance(first.get("confidence"), (int, float)): + confidence = float(first["confidence"]) + else: + text = str(first).strip() + return ASRResult( + text=text, + confidence=confidence, + model=self.model_path, + device=self.device, + ) diff --git a/extensions/assistive_harness/config/skills.example.yaml b/extensions/assistive_harness/config/skills.example.yaml new file mode 100644 index 0000000..4e589ff --- /dev/null +++ b/extensions/assistive_harness/config/skills.example.yaml @@ -0,0 +1,128 @@ +version: 2 +default_skill: idle_chat + +harness: + enabled: false + host: 127.0.0.1 + port: 8021 + allow_test_injection: false + +asr: + engine: funasr + model_path: "" + device: cpu + rms_threshold: 0.012 + min_speech_ms: 180 + end_silence_ms: 450 + max_utterance_ms: 8000 + preroll_ms: 200 + +echo_guard: + window_ms: 20000 + similarity_threshold: 0.86 + +cv: + mode: shadow + max_fps: 1 + queue_size: 1 + inference_timeout_ms: 2000 + providers: + noop: + type: noop + yolo_onnx: + type: yolo_onnx + # Resolved relative to this YAML. Keep weights outside the Core so the + # same plugin can be reused by the later OpenGlass Adapter. + model_path: ../../../../OmniHarness/mini_omni_harness/models/yolo26n.onnx + device: cpu + confidence: 0.25 + image_size: 640 + +control: + cooldown_ms: 1200 + stop_speech: + # `鍚屼竴涓媊 / `绛変竴(涓)涓媊 are bounded aliases observed from the local + # FunASR model when the user clearly says the short command `鍋滀竴涓媊. + # They are command-only aliases, not arbitrary substring anchors. + phrases: [鍋滀竴涓, 鍚屼竴涓, 绛変竴涓, 绛変竴涓涓, 鍒浜, 闂槾, 鍋滄鎾姤, 瀹夐潤] + # Literal substring protocol: any transcript containing this anchor stops. + embedded_phrases: [鍋滀竴涓媇 + resume_speech: + # Literal substring protocol: any transcript containing this anchor resumes. + embedded_phrases: [鎭㈠瀵硅瘽] + reset_session: + phrases: [閲嶆柊寮濮媇 + # Literal substring protocol for real voice RESET. Product deployments can + # later prepend a wake name (for example "涔愬閲嶆柊寮濮") without changing + # the deterministic router. + embedded_phrases: [閲嶆柊寮濮媇 + return_to_chat: + phrases: [鍥炲埌鑱婂ぉ, 鍥炲埌鏅氳亰澶, 閫鍑烘妧鑳, 鏅氳亰澶, 鎭㈠鏅氳亰澶 + cancel_skill: + phrases: [鍙栨秷浠诲姟, 涓嶆壘浜, 涓嶈浜哴 + +skills: + idle_chat: + enabled: true + prompt_file: ../prompts/idle_chat_zh.txt + requires_session_restart: true + cooldown_ms: 1200 + cv_mode: disabled + cv_provider: noop + activation_phrases: [鏅氳亰澶, 鍥炲埌鑱婂ぉ] + + find_object: + enabled: true + description: 浣跨敤绗竴瑙嗚鐢婚潰瀵绘壘璇煶涓寚瀹氱殑鐗╀綋 + prompt_source: AAAI_SI/C1_OBJECT_FINDING + prompt_file: ../prompts/find_object_zh.txt + requires_session_restart: true + cooldown_ms: 1500 + cv_mode: shadow + cv_provider: yolo_onnx + task_trigger: '璇风珛鍗虫牴鎹綋鍓嶇敾闈㈠鎵锯渰{target}}鈥濓紝鍙洖绛斿畠鐨勪綅缃紱濡傛灉娌$湅鍒板氨璇存病鐪嬪埌銆' + slot_schema: + target: {type: string, required: true} + activation_patterns: + - "甯垜鎵(?:涓涓)?(?:鎴戠殑)?(?P.+)" + - "甯垜鎵(?P.+)" + - "(?P.+)鍦ㄥ摢" + - "(?P.+)鍦ㄥ摢閲" + - "鐪嬪埌(?P.+)浜嗗悧" + - "鏈夋病鏈(?P.+)" + + read_text: + enabled: true + description: 璇诲彇褰撳墠鐢婚潰涓竻鏅板彲瑙佺殑鏂囧瓧 + prompt_source: AAAI_SI/C1_TEXT_READING + prompt_file: ../prompts/read_text_zh.txt + requires_session_restart: true + cooldown_ms: 1500 + cv_mode: shadow + cv_provider: noop + task_trigger: 璇风珛鍗宠鍙栧綋鍓嶇敾闈腑鏈鏄庢樉鐨勬枃瀛楋紝鍙鐪嬪埌鐨勫唴瀹广 + activation_phrases: [璇讳竴涓, 蹇典竴涓, 甯垜璇讳竴涓, 甯垜璇嗗瓧, 甯垜璇诲瓧, 涓婇潰鍐欎簡浠涔, 杩欐槸浠涔堝瓧, 璇绘枃瀛梋 + + describe_scene: + enabled: true + prompt_file: ../prompts/describe_scene_zh.txt + requires_session_restart: true + cooldown_ms: 1500 + cv_mode: shadow + cv_provider: noop + task_trigger: 璇风珛鍗崇敤涓涓ゅ彞璇濈畝鐭弿杩板綋鍓嶇敾闈€ + activation_phrases: [鎻忚堪涓涓, 甯垜鎻忚堪涓涓, 甯垜鎻忚堪鍦烘櫙, 鍓嶉潰鏈変粈涔, 鐪嬬湅鍛ㄥ洿, 鐢婚潰閲屾湁浠涔圿 + + obstacle_avoidance: + # Experimental: enabled only for stationary, supervised browser validation. + # It is not a mobility-safety guarantee and must not be used for navigation. + enabled: true + description: 璇嗗埆鍓嶆柟鍙闅滅骞剁粰鍑轰繚瀹堟彁绀猴紙瀹為獙锛 + prompt_source: AAAI_SI/C1_OBSTACLE_AVOID_DIRECT_V1 + prompt_file: ../prompts/obstacle_avoidance_zh.txt + requires_session_restart: true + cooldown_ms: 1500 + cv_mode: shadow + cv_provider: noop + task_trigger: 璇风珛鍗冲垽鏂綋鍓嶇敾闈㈠墠鏂规槸鍚︽湁鏄庢樉闅滅锛屽苟缁欏嚭绠鐭佷繚瀹堢殑瀹夊叏鎻愮ず銆 + activation_phrases: [甯垜閬块殰, 鍓嶉潰鏈夐殰纰嶅悧, 鍓嶉潰鏈夋病鏈夐殰纰, 鐪嬩竴涓嬪墠鏂归殰纰, 鍓嶉潰瀹夊叏鍚梋 diff --git a/extensions/assistive_harness/cv/README.md b/extensions/assistive_harness/cv/README.md new file mode 100644 index 0000000..6c986c2 --- /dev/null +++ b/extensions/assistive_harness/cv/README.md @@ -0,0 +1,171 @@ +# Lightweight CV provider framework (V1) + +This directory is the transport-neutral CV shadow path used by the Assistive +Harness. Browser and future OpenGlass Adapters only provide JPEG frames. They do +not call YOLO, OCR, depth estimation, or control rules directly. + +## V1 data flow + +```text +Browser / ESP32 Adapter + | + | FrameEnvelope(JPEG, timestamp, skill_id, slots) + v +CVPipeline.submit() # returns immediately + | + | capacity-one latest-frame queue (old queued frame is dropped) + v +single background worker + | + | provider.analyze() in a dedicated CPU thread + v +CVObservation -> cv_events.jsonl # shadow only; no control decision +``` + +The Router, state machine, Session Adapter, and MiniCPM audio/video path do not +depend on a concrete CV model. + +## Guarantees and deliberate limits + +- `submit()` never waits for inference. +- The queue is bounded and latest-frame wins; overload cannot grow memory + without limit. +- Only one inference call runs at a time per client pipeline. +- Provider errors become `CVObservation.error` and never escape to the control + socket. +- A timeout is recorded without blocking STOP/RESET. Python cannot kill an + already-running native ONNX call safely, so that one worker lane remains + occupied until the native call returns; new queued frames continue to + collapse to the latest frame. +- V1 accepts only `disabled` and `shadow`. Shadow observations never switch a + Skill, suppress MiniCPM, or speak to the user. +- Temporal voting, result fusion, navigation decisions, and accuracy tuning are + intentionally outside V1. + +## Configuration + +Providers are declared once under `cv.providers`. Each Skill selects one with +`cv_provider`: + +```yaml +cv: + max_fps: 1 + queue_size: 1 + inference_timeout_ms: 2000 + providers: + noop: + type: noop + yolo_onnx: + type: yolo_onnx + model_path: path/to/yolo.onnx + device: cpu + confidence: 0.25 + image_size: 640 + +skills: + find_object: + cv_mode: shadow + cv_provider: yolo_onnx +``` + +Relative model paths are resolved against the YAML directory. Weights are not +owned by the Core. The current local sample points to the previously validated +`OmniHarness/mini_omni_harness/models/yolo26n.onnx` file. + +## Provider contract + +A provider is synchronous. `CVPipeline` is responsible for running it outside +the asyncio/control loop: + +Provider instances are shared so model weights load only once. If a backend is +not safe for concurrent calls from multiple connected clients, the provider +must protect that backend with its own lock, as `YoloOnnxProvider` does. + +```python +from typing import Any +from extensions.assistive_harness.cv.base import CVObservation + + +class OcrOnnxProvider: + def __init__(self, model_path): + self.model_path = model_path + self._session = None # lazy-load on first frame + + def analyze( + self, + frame: bytes | None, + frame_id: str, + timestamp_ms: float, + skill_id: str, + slots: dict[str, Any], + ) -> CVObservation: + if not frame: + raise ValueError("OCR requires a JPEG frame") + # 1. Decode JPEG. + # 2. Lazy-load and run the local model. + # 3. Return JSON-serializable values only. + return CVObservation( + frame_id=frame_id, + timestamp_ms=timestamp_ms, + skill_id=skill_id, + provider="ocr_onnx", + values={ + "status": "ok", + "text": "recognized text", + "regions": [], + "latency_ms": 12.3, + "model": self.model_path.name, + "device": "cpu", + }, + ) +``` + +To add OCR: + +1. Put the implementation in `cv/ocr_onnx.py` and keep model-specific imports + inside that plugin. +2. Add one `ocr_onnx` factory branch in `build_provider_registry()` in + `cv/registry.py`. +3. Add its config under `cv.providers`. +4. Set `read_text.cv_provider: ocr_onnx`; leave `cv_mode: shadow` during tests. +5. Add a blank/synthetic inference smoke plus one recorded real-frame test. + +No changes are required in `router.py`, `state_machine.py`, the browser Adapter, +or the future ESP32 Frame Adapter. + +## Observation schema + +Every plugin produces the same envelope: + +```json +{ + "frame_id": "browser_123", + "timestamp_ms": 123.0, + "skill_id": "find_object", + "provider": "shadow:yolo_onnx", + "values": { + "status": "ok", + "found": true, + "detections": [], + "latency_ms": 31.1, + "pipeline_latency_ms": 42.0 + }, + "error": null +} +``` + +Plugin-specific data belongs under `values`; the outer keys stay stable for +logging, replay, and later OpenGlass migration. + +## Validation + +```powershell +python -m unittest ` + extensions.assistive_harness.tests.test_core ` + extensions.assistive_harness.tests.test_cv_yolo +``` + +The tests cover failure isolation, latest-frame dropping, timeout reporting, +non-blocking submission, and a real ONNX Runtime YOLO inference. A browser +acceptance run should additionally verify that `find_object` produces +`provider=shadow:yolo_onnx` records while STOP/RESET remain responsive. diff --git a/extensions/assistive_harness/cv/__init__.py b/extensions/assistive_harness/cv/__init__.py new file mode 100644 index 0000000..dec2042 --- /dev/null +++ b/extensions/assistive_harness/cv/__init__.py @@ -0,0 +1,14 @@ +from .base import CVObservation, FrameEnvelope, PerceptionProvider +from .noop import NoOpCVProvider, ShadowCVProvider +from .pipeline import CVPipeline +from .registry import CVProviderRegistry + +__all__ = [ + "CVObservation", + "CVPipeline", + "CVProviderRegistry", + "FrameEnvelope", + "NoOpCVProvider", + "PerceptionProvider", + "ShadowCVProvider", +] diff --git a/extensions/assistive_harness/cv/base.py b/extensions/assistive_harness/cv/base.py new file mode 100644 index 0000000..5c1e153 --- /dev/null +++ b/extensions/assistive_harness/cv/base.py @@ -0,0 +1,44 @@ +from __future__ import annotations + +from dataclasses import asdict, dataclass, field +from typing import Any, Protocol + + +@dataclass(slots=True) +class FrameEnvelope: + """Transport-neutral frame handed from an Adapter to the CV worker.""" + + frame: bytes | None + frame_id: str + timestamp_ms: float + skill_id: str + slots: dict[str, Any] = field(default_factory=dict) + mode: str = "shadow" + provider_id: str = "noop" + + +@dataclass(slots=True) +class CVObservation: + frame_id: str + timestamp_ms: float + skill_id: str + provider: str + values: dict[str, Any] = field(default_factory=dict) + error: str | None = None + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +class PerceptionProvider(Protocol): + """Synchronous plugin contract; CVPipeline always calls it off-loop.""" + + def analyze( + self, + frame: bytes | None, + frame_id: str, + timestamp_ms: float, + skill_id: str, + slots: dict[str, Any], + ) -> CVObservation: + ... diff --git a/extensions/assistive_harness/cv/noop.py b/extensions/assistive_harness/cv/noop.py new file mode 100644 index 0000000..1770c62 --- /dev/null +++ b/extensions/assistive_harness/cv/noop.py @@ -0,0 +1,60 @@ +from __future__ import annotations + +from typing import Any + +from .base import CVObservation, PerceptionProvider + + +class NoOpCVProvider: + def analyze( + self, + frame: bytes | None, + frame_id: str, + timestamp_ms: float, + skill_id: str, + slots: dict[str, Any], + ) -> CVObservation: + del frame, slots + return CVObservation( + frame_id=frame_id, + timestamp_ms=timestamp_ms, + skill_id=skill_id, + provider="noop", + values={}, + ) + + +class ShadowCVProvider: + """Failure-isolated observer. Its result is never a control decision.""" + + def __init__( + self, + provider: PerceptionProvider | None = None, + provider_id: str | None = None, + ): + self.provider = provider or NoOpCVProvider() + self.provider_id = provider_id or type(self.provider).__name__ + + def analyze( + self, + frame: bytes | None, + frame_id: str, + timestamp_ms: float, + skill_id: str, + slots: dict[str, Any], + ) -> CVObservation: + try: + result = self.provider.analyze( + frame, frame_id, timestamp_ms, skill_id, slots + ) + result.provider = f"shadow:{result.provider}" + return result + except Exception as exc: # shadow failures must not escape + return CVObservation( + frame_id=frame_id, + timestamp_ms=timestamp_ms, + skill_id=skill_id, + provider=f"shadow:{self.provider_id}", + values={}, + error=f"{type(exc).__name__}: {exc}", + ) diff --git a/extensions/assistive_harness/cv/pipeline.py b/extensions/assistive_harness/cv/pipeline.py new file mode 100644 index 0000000..f17f9cf --- /dev/null +++ b/extensions/assistive_harness/cv/pipeline.py @@ -0,0 +1,159 @@ +from __future__ import annotations + +import asyncio +import time +from concurrent.futures import ThreadPoolExecutor +from typing import Callable + +from .base import CVObservation, FrameEnvelope +from .registry import CVProviderRegistry + + +ObservationCallback = Callable[[CVObservation], None] +MetricCallback = Callable[[str, float], None] + + +class CVPipeline: + """Latest-frame-only worker that keeps CV inference off the control loop.""" + + def __init__( + self, + providers: CVProviderRegistry, + *, + on_observation: ObservationCallback, + on_metric: MetricCallback | None = None, + queue_size: int = 1, + inference_timeout_ms: float = 2000.0, + worker_name: str = "assistive-cv", + ) -> None: + if int(queue_size) != 1: + raise ValueError("CV V1 requires queue_size=1 (latest-frame-only)") + self.providers = providers + self.on_observation = on_observation + self.on_metric = on_metric + self.queue: asyncio.Queue[FrameEnvelope] = asyncio.Queue(maxsize=1) + self.inference_timeout_ms = max(1.0, float(inference_timeout_ms)) + self.executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix=worker_name) + self.worker_task: asyncio.Task[None] | None = None + self.closed = False + self.submitted_frames = 0 + self.processed_frames = 0 + self.dropped_frames = 0 + self.timeout_count = 0 + + def submit(self, envelope: FrameEnvelope) -> bool: + """Enqueue without awaiting inference; replace the oldest queued frame.""" + if self.closed or envelope.mode == "disabled": + return False + self._ensure_worker() + if self.queue.full(): + try: + self.queue.get_nowait() + self.queue.task_done() + self.dropped_frames += 1 + self._metric("cv_frames_dropped", 1.0) + except asyncio.QueueEmpty: + pass + self.queue.put_nowait(envelope) + self.submitted_frames += 1 + return True + + def snapshot(self) -> dict[str, int | bool]: + return { + "closed": self.closed, + "queued_frames": self.queue.qsize(), + "submitted_frames": self.submitted_frames, + "processed_frames": self.processed_frames, + "dropped_frames": self.dropped_frames, + "timeout_count": self.timeout_count, + } + + async def close(self) -> None: + if self.closed: + return + self.closed = True + if self.worker_task is not None: + self.worker_task.cancel() + await asyncio.gather(self.worker_task, return_exceptions=True) + self.executor.shutdown(wait=False, cancel_futures=True) + + def _ensure_worker(self) -> None: + if self.worker_task is None or self.worker_task.done(): + self.worker_task = asyncio.create_task(self._run()) + + async def _run(self) -> None: + loop = asyncio.get_running_loop() + while True: + envelope = await self.queue.get() + started = time.perf_counter() + future = loop.run_in_executor(self.executor, self.providers.analyze, envelope) + try: + observation = await asyncio.wait_for( + asyncio.shield(future), self.inference_timeout_ms / 1000.0 + ) + except asyncio.TimeoutError: + self.timeout_count += 1 + self._metric("cv_inference_timeout", 1.0) + self._emit( + CVObservation( + frame_id=envelope.frame_id, + timestamp_ms=envelope.timestamp_ms, + skill_id=envelope.skill_id, + provider=f"{envelope.mode}:{envelope.provider_id}", + values={ + "status": "timeout", + "timeout_ms": self.inference_timeout_ms, + }, + error=( + "TimeoutError: CV inference exceeded " + f"{self.inference_timeout_ms:.0f} ms" + ), + ) + ) + # Python cannot safely kill an in-flight native inference call. + # Keep this one-worker lane occupied until it exits so timeouts + # never create concurrent calls or an unbounded executor queue. + try: + await future + except Exception: + pass + except asyncio.CancelledError: + raise + except Exception as exc: + self._emit( + CVObservation( + frame_id=envelope.frame_id, + timestamp_ms=envelope.timestamp_ms, + skill_id=envelope.skill_id, + provider=f"{envelope.mode}:{envelope.provider_id}", + values={"status": "error"}, + error=f"{type(exc).__name__}: {exc}", + ) + ) + else: + observation.values.setdefault( + "pipeline_latency_ms", round((time.perf_counter() - started) * 1000, 1) + ) + self.processed_frames += 1 + self._metric( + "cv_pipeline_latency_ms", + float(observation.values["pipeline_latency_ms"]), + ) + self._emit(observation) + finally: + self.queue.task_done() + + def _emit(self, observation: CVObservation) -> None: + try: + self.on_observation(observation) + except Exception: + # Telemetry/consumer failures are also outside the control path. + pass + + def _metric(self, name: str, value: float) -> None: + if self.on_metric is None: + return + try: + self.on_metric(name, value) + except Exception: + pass diff --git a/extensions/assistive_harness/cv/registry.py b/extensions/assistive_harness/cv/registry.py new file mode 100644 index 0000000..dbe3b4e --- /dev/null +++ b/extensions/assistive_harness/cv/registry.py @@ -0,0 +1,89 @@ +from __future__ import annotations + +from pathlib import Path +from typing import Any + +from .base import CVObservation, FrameEnvelope, PerceptionProvider +from .noop import NoOpCVProvider, ShadowCVProvider + + +class CVProviderRegistry: + """Small explicit registry shared by browser and future device Adapters.""" + + def __init__(self, providers: dict[str, PerceptionProvider] | None = None): + self.providers: dict[str, PerceptionProvider] = { + "noop": NoOpCVProvider(), + **(providers or {}), + } + self._shadow = { + provider_id: ShadowCVProvider(provider, provider_id) + for provider_id, provider in self.providers.items() + } + + def register(self, provider_id: str, provider: PerceptionProvider) -> None: + normalized = provider_id.strip() + if not normalized: + raise ValueError("provider_id cannot be empty") + self.providers[normalized] = provider + self._shadow[normalized] = ShadowCVProvider(provider, normalized) + + def analyze(self, envelope: FrameEnvelope) -> CVObservation: + if envelope.mode != "shadow": + return CVObservation( + frame_id=envelope.frame_id, + timestamp_ms=envelope.timestamp_ms, + skill_id=envelope.skill_id, + provider=f"{envelope.mode}:{envelope.provider_id}", + values={"status": "skipped", "reason": "unsupported_cv_mode"}, + error=f"Unsupported cv_mode: {envelope.mode}", + ) + provider = self._shadow.get(envelope.provider_id) + if provider is None: + return CVObservation( + frame_id=envelope.frame_id, + timestamp_ms=envelope.timestamp_ms, + skill_id=envelope.skill_id, + provider=f"shadow:{envelope.provider_id}", + values={"status": "skipped", "reason": "unknown_provider"}, + error=f"Unknown CV provider: {envelope.provider_id}", + ) + return provider.analyze( + envelope.frame, + envelope.frame_id, + envelope.timestamp_ms, + envelope.skill_id, + envelope.slots, + ) + + +def build_provider_registry( + config: dict[str, Any], *, config_dir: Path +) -> CVProviderRegistry: + registry = CVProviderRegistry() + providers = dict((config.get("cv") or {}).get("providers") or {}) + for provider_id, raw_spec in providers.items(): + spec = dict(raw_spec or {}) + provider_type = str(spec.get("type") or provider_id) + if provider_type == "noop": + registry.register(str(provider_id), NoOpCVProvider()) + continue + if provider_type == "yolo_onnx": + from .yolo_onnx import YoloOnnxProvider + + raw_path = Path(str(spec.get("model_path") or "")) + model_path = raw_path if raw_path.is_absolute() else config_dir / raw_path + registry.register( + str(provider_id), + YoloOnnxProvider( + model_path.resolve(), + device=str(spec.get("device") or "cpu"), + confidence=float(spec.get("confidence", 0.25)), + image_size=int(spec.get("image_size", 640)), + target_aliases=dict(spec.get("target_aliases") or {}), + ), + ) + continue + raise ValueError( + f"Unsupported CV provider type {provider_type!r} for {provider_id!r}" + ) + return registry diff --git a/extensions/assistive_harness/cv/requirements-cv.txt b/extensions/assistive_harness/cv/requirements-cv.txt new file mode 100644 index 0000000..72c2319 --- /dev/null +++ b/extensions/assistive_harness/cv/requirements-cv.txt @@ -0,0 +1,2 @@ +opencv-python-headless +onnxruntime==1.21.0 diff --git a/extensions/assistive_harness/cv/yolo_onnx.py b/extensions/assistive_harness/cv/yolo_onnx.py new file mode 100644 index 0000000..01ca1c8 --- /dev/null +++ b/extensions/assistive_harness/cv/yolo_onnx.py @@ -0,0 +1,220 @@ +from __future__ import annotations + +import threading +import time +from pathlib import Path +from typing import Any + +import numpy as np + +from .base import CVObservation + + +COCO_NAMES = ( + "person", "bicycle", "car", "motorcycle", "airplane", "bus", "train", + "truck", "boat", "traffic light", "fire hydrant", "stop sign", + "parking meter", "bench", "bird", "cat", "dog", "horse", "sheep", + "cow", "elephant", "bear", "zebra", "giraffe", "backpack", "umbrella", + "handbag", "tie", "suitcase", "frisbee", "skis", "snowboard", + "sports ball", "kite", "baseball bat", "baseball glove", "skateboard", + "surfboard", "tennis racket", "bottle", "wine glass", "cup", "fork", + "knife", "spoon", "bowl", "banana", "apple", "sandwich", "orange", + "broccoli", "carrot", "hot dog", "pizza", "donut", "cake", "chair", + "couch", "potted plant", "bed", "dining table", "toilet", "tv", "laptop", + "mouse", "remote", "keyboard", "cell phone", "microwave", "oven", + "toaster", "sink", "refrigerator", "book", "clock", "vase", "scissors", + "teddy bear", "hair drier", "toothbrush", +) + +DEFAULT_TARGET_ALIASES = { + "鎵嬫満": "cell phone", "鐢佃瘽": "cell phone", "鏅鸿兘鎵嬫満": "cell phone", + "phone": "cell phone", "cellphone": "cell phone", "cell phone": "cell phone", + "涔": "book", "涔︽湰": "book", "鍥句功": "book", "book": "book", + "鏉瓙": "cup", "姘存澂": "cup", "鑼舵澂": "cup", "cup": "cup", + "鐡跺瓙": "bottle", "姘寸摱": "bottle", "bottle": "bottle", + "妞呭瓙": "chair", "chair": "chair", "浜": "person", "琛屼汉": "person", + "person": "person", "鐢佃剳": "laptop", "绗旇鏈數鑴": "laptop", + "laptop": "laptop", "閬ユ帶鍣": "remote", "remote": "remote", + "閿洏": "keyboard", "keyboard": "keyboard", "榧犳爣": "mouse", + "mouse": "mouse", "鑳屽寘": "backpack", "涔﹀寘": "backpack", + "backpack": "backpack", +} + + +class YoloOnnxProvider: + """CPU-first YOLO ONNX reference plugin with no control-side effects.""" + + def __init__( + self, + model_path: Path, + *, + device: str = "cpu", + confidence: float = 0.25, + image_size: int = 640, + target_aliases: dict[str, str] | None = None, + ) -> None: + if device.lower() != "cpu": + raise ValueError("CV V1 only accepts device=cpu") + self.model_path = Path(model_path).resolve() + self.device = "cpu" + self.confidence = float(confidence) + self.image_size = int(image_size) + self.target_aliases = { + **DEFAULT_TARGET_ALIASES, + **{str(k).lower(): str(v) for k, v in (target_aliases or {}).items()}, + } + self._session: Any = None + self._input_name: str | None = None + self._cv2: Any = None + self._lock = threading.Lock() + + def analyze( + self, + frame: bytes | None, + frame_id: str, + timestamp_ms: float, + skill_id: str, + slots: dict[str, Any], + ) -> CVObservation: + if not frame: + raise ValueError("YOLO requires a non-empty JPEG frame") + self._load() + target = str(slots.get("target") or "").strip() + target_label = self._resolve_target(target) if target else None + if target and target_label is None: + return CVObservation( + frame_id=frame_id, + timestamp_ms=timestamp_ms, + skill_id=skill_id, + provider="yolo_onnx", + values={ + "status": "skipped", + "reason": "unsupported_target", + "target": target, + }, + ) + + encoded = np.frombuffer(frame, dtype=np.uint8) + image = self._cv2.imdecode(encoded, self._cv2.IMREAD_COLOR) + if image is None: + raise ValueError("YOLO could not decode the JPEG frame") + height, width = image.shape[:2] + started = time.perf_counter() + tensor, scale, pad_x, pad_y = self._prepare_input(image) + with self._lock: + output = self._session.run(None, {self._input_name: tensor})[0] + latency_ms = (time.perf_counter() - started) * 1000.0 + if output.ndim != 3 or output.shape[0] != 1 or output.shape[2] != 6: + raise ValueError(f"Unexpected YOLO output shape: {output.shape}") + + target_id = COCO_NAMES.index(target_label) if target_label else None + detections: list[dict[str, Any]] = [] + for row in output[0]: + x1, y1, x2, y2, confidence, raw_class = row + class_id = round(float(raw_class)) + if float(confidence) < self.confidence or not 0 <= class_id < len(COCO_NAMES): + continue + if target_id is not None and class_id != target_id: + continue + left = max(0.0, min(width, (float(x1) - pad_x) / scale)) + top = max(0.0, min(height, (float(y1) - pad_y) / scale)) + right = max(0.0, min(width, (float(x2) - pad_x) / scale)) + bottom = max(0.0, min(height, (float(y2) - pad_y) / scale)) + if right <= left or bottom <= top: + continue + center_x = ((left + right) / 2.0) / width + center_y = ((top + bottom) / 2.0) / height + detections.append( + { + "label": COCO_NAMES[class_id], + "confidence": round(float(confidence), 4), + "bbox_xyxy": [round(left), round(top), round(right), round(bottom)], + "bbox_normalized": [ + round(left / width, 4), round(top / height, 4), + round(right / width, 4), round(bottom / height, 4), + ], + "center_normalized": [round(center_x, 4), round(center_y, 4)], + "position": self._position(center_x, center_y), + } + ) + detections.sort(key=lambda item: float(item["confidence"]), reverse=True) + return CVObservation( + frame_id=frame_id, + timestamp_ms=timestamp_ms, + skill_id=skill_id, + provider="yolo_onnx", + values={ + "status": "ok", + "found": bool(detections), + "target": target or None, + "canonical_label": target_label, + "best_detection": detections[0] if detections else None, + "detections": detections, + "image": {"width": width, "height": height}, + "latency_ms": round(latency_ms, 1), + "frame_age_ms": round(max(0.0, time.time() * 1000.0 - timestamp_ms), 1), + "model": self.model_path.name, + "device": self.device, + }, + ) + + def _load(self) -> None: + if self._session is not None: + return + if not self.model_path.is_file(): + raise FileNotFoundError(f"YOLO weights not found: {self.model_path}") + import cv2 + import onnxruntime as ort + + session = ort.InferenceSession( + str(self.model_path), providers=["CPUExecutionProvider"] + ) + model_input = session.get_inputs()[0] + if model_input.shape[-2:] != [self.image_size, self.image_size]: + raise ValueError( + f"YOLO input is {model_input.shape[-2:]}, expected " + f"[{self.image_size}, {self.image_size}]" + ) + self._cv2 = cv2 + self._session = session + self._input_name = model_input.name + + def _prepare_input( + self, image: np.ndarray + ) -> tuple[np.ndarray, float, float, float]: + height, width = image.shape[:2] + scale = min(self.image_size / width, self.image_size / height) + resized_width, resized_height = round(width * scale), round(height * scale) + resized = self._cv2.resize(image, (resized_width, resized_height)) + pad_x = (self.image_size - resized_width) / 2.0 + pad_y = (self.image_size - resized_height) / 2.0 + left, top = round(pad_x - 0.1), round(pad_y - 0.1) + right = self.image_size - resized_width - left + bottom = self.image_size - resized_height - top + padded = self._cv2.copyMakeBorder( + resized, top, bottom, left, right, + self._cv2.BORDER_CONSTANT, value=(114, 114, 114), + ) + rgb = self._cv2.cvtColor(padded, self._cv2.COLOR_BGR2RGB) + tensor = np.ascontiguousarray(rgb.transpose(2, 0, 1), dtype=np.float32) + return np.expand_dims(tensor / 255.0, axis=0), scale, float(left), float(top) + + def _resolve_target(self, target: str) -> str | None: + normalized = target.lower().replace(" ", "") + for alias in sorted(self.target_aliases, key=len, reverse=True): + if alias.lower().replace(" ", "") in normalized: + label = self.target_aliases[alias] + return label if label in COCO_NAMES else None + return None + + @staticmethod + def _position(center_x: float, center_y: float) -> str: + horizontal = "宸" if center_x < 1 / 3 else "鍙" if center_x > 2 / 3 else "涓" + vertical = "涓" if center_y < 1 / 3 else "涓" if center_y > 2 / 3 else "涓" + return { + ("宸", "涓"): "宸︿笂鏂", ("涓", "涓"): "姝d笂鏂", + ("鍙", "涓"): "鍙充笂鏂", ("宸", "涓"): "宸︿晶", + ("涓", "涓"): "涓ぎ", ("鍙", "涓"): "鍙充晶", + ("宸", "涓"): "宸︿笅鏂", ("涓", "涓"): "姝d笅鏂", + ("鍙", "涓"): "鍙充笅鏂", + }[(horizontal, vertical)] diff --git a/extensions/assistive_harness/echo_guard.py b/extensions/assistive_harness/echo_guard.py new file mode 100644 index 0000000..fc3d25e --- /dev/null +++ b/extensions/assistive_harness/echo_guard.py @@ -0,0 +1,75 @@ +from __future__ import annotations + +import time +from collections import deque +from dataclasses import dataclass +from difflib import SequenceMatcher + +from .router import normalize_text +from .schemas import ControlIntent + + +@dataclass(frozen=True, slots=True) +class EchoDecision: + allow: bool + reason: str + similarity: float = 0.0 + + +class EchoGuard: + def __init__(self, window_ms: int = 20_000, similarity_threshold: float = 0.86): + self.window_ms = int(window_ms) + self.similarity_threshold = float(similarity_threshold) + self._model_text: deque[tuple[float, str]] = deque() + self.ai_speaking = False + + def note_model_text(self, text: str, at_ms: float | None = None) -> None: + normalized = normalize_text(text) + if not normalized: + return + now = at_ms if at_ms is not None else time.time() * 1000 + self._model_text.append((now, normalized)) + self._prune(now) + + def set_ai_speaking(self, active: bool) -> None: + self.ai_speaking = bool(active) + + def _prune(self, now_ms: float) -> None: + cutoff = now_ms - self.window_ms + while self._model_text and self._model_text[0][0] < cutoff: + self._model_text.popleft() + + def evaluate( + self, + utterance: str, + intent: ControlIntent, + at_ms: float | None = None, + ) -> EchoDecision: + now = at_ms if at_ms is not None else time.time() * 1000 + self._prune(now) + normalized = normalize_text(utterance) + if len(normalized) < 2: + return EchoDecision(False, "too_short") + + best = 0.0 + for _, model_text in self._model_text: + if normalized in model_text or model_text in normalized: + best = 1.0 + break + best = max(best, SequenceMatcher(None, normalized, model_text).ratio()) + if best >= self.similarity_threshold: + return EchoDecision(False, "matches_recent_model_echo", best) + + if self.ai_speaking and intent not in { + ControlIntent.STOP_SPEECH, + ControlIntent.RESUME_SPEECH, + ControlIntent.RESET_SESSION, + ControlIntent.CANCEL_SKILL, + ControlIntent.RETURN_TO_CHAT, + # A positively routed Skill command is also a control-plane + # interruption. Recent-model-text similarity is checked above, + # so model echo is still rejected before this exception applies. + ControlIntent.ACTIVATE_SKILL, + }: + return EchoDecision(False, "ordinary_skill_suppressed_while_ai_speaking", best) + return EchoDecision(True, "allowed", best) diff --git a/extensions/assistive_harness/model_log.py b/extensions/assistive_harness/model_log.py new file mode 100644 index 0000000..43045e4 --- /dev/null +++ b/extensions/assistive_harness/model_log.py @@ -0,0 +1,72 @@ +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any + + +@dataclass(slots=True) +class ModelTurnAccumulator: + """Aggregate streamed Gateway fragments into auditable model turns.""" + + turn_index: int = 0 + _key: tuple[Any, ...] | None = None + _text: list[str] = field(default_factory=list) + _audio_ms: float = 0.0 + _metadata: dict[str, Any] = field(default_factory=dict) + + def _flush(self) -> dict[str, Any] | None: + if self._key is None: + return None + self.turn_index += 1 + record = { + "type": "model.turn", + "turn_index": self.turn_index, + **self._metadata, + "text": "".join(self._text), + "audio_ms": round(self._audio_ms, 1), + } + self._key = None + self._text.clear() + self._audio_ms = 0.0 + self._metadata = {} + return record + + def feed(self, message: dict[str, Any]) -> list[dict[str, Any]]: + state = str(message.get("state") or "unknown") + key = ( + message.get("session_id"), + int(message.get("generation") or 0), + str(message.get("skill_id") or ""), + state, + ) + completed: list[dict[str, Any]] = [] + if self._key is not None and key != self._key: + record = self._flush() + if record is not None: + completed.append(record) + + text = str(message.get("text") or "") + audio_ms = float(message.get("audio_ms") or 0.0) + end_of_turn = bool(message.get("end_of_turn")) + if self._key is None and (text or audio_ms > 0): + self._key = key + self._metadata = { + "role": "assistant" if state == "speak" else "listen", + "state": state, + "session_id": message.get("session_id"), + "generation": int(message.get("generation") or 0), + "skill_id": str(message.get("skill_id") or ""), + "slots": dict(message.get("slots") or {}), + } + if self._key is not None: + if text: + self._text.append(text) + self._audio_ms += max(0.0, audio_ms) + if end_of_turn: + record = self._flush() + if record is not None: + completed.append(record) + return completed + + def flush(self) -> dict[str, Any] | None: + return self._flush() diff --git a/extensions/assistive_harness/package_phase_a.ps1 b/extensions/assistive_harness/package_phase_a.ps1 new file mode 100644 index 0000000..8264580 --- /dev/null +++ b/extensions/assistive_harness/package_phase_a.ps1 @@ -0,0 +1,69 @@ +param( + [string]$RepoRoot = (Resolve-Path (Join-Path $PSScriptRoot '..\..')).Path, + [string]$Date = '2026-08-11' +) + +$ErrorActionPreference = 'Stop' +$deliverables = Join-Path $RepoRoot 'deliverables' +$packageName = "VOICE_SKILL_HARNESS_PHASE_A_CODEX_RETURN_$Date" +$stage = Join-Path $deliverables $packageName +$zipPath = Join-Path $deliverables "$packageName.zip" + +$resolvedRepo = (Resolve-Path $RepoRoot).Path.TrimEnd('\') +New-Item -ItemType Directory -Force -Path $deliverables | Out-Null +$resolvedDeliverables = (Resolve-Path $deliverables).Path +if (-not $resolvedDeliverables.StartsWith($resolvedRepo, [StringComparison]::OrdinalIgnoreCase)) { + throw "Unsafe delivery target: $resolvedDeliverables" +} +if (Test-Path $stage) { Remove-Item -LiteralPath $stage -Recurse -Force } +if (Test-Path $zipPath) { Remove-Item -LiteralPath $zipPath -Force } +New-Item -ItemType Directory -Force -Path $stage | Out-Null + +function Copy-TreeFiltered([string]$Source, [string]$Destination) { + Get-ChildItem -LiteralPath $Source -Recurse -File | Where-Object { + $_.FullName -notmatch '[\\/]runs[\\/]' -and + $_.FullName -notmatch '[\\/]__pycache__[\\/]' -and + $_.FullName -notmatch '[\\/]test_artifacts[\\/]' -and + $_.Extension -ne '.pyc' + } | ForEach-Object { + $relative = $_.FullName.Substring($Source.TrimEnd('\').Length).TrimStart('\') + $target = Join-Path $Destination $relative + New-Item -ItemType Directory -Force -Path (Split-Path $target) | Out-Null + Copy-Item -LiteralPath $_.FullName -Destination $target + } +} + +Copy-TreeFiltered (Join-Path $RepoRoot 'extensions\assistive_harness') (Join-Path $stage 'extensions\assistive_harness') +Copy-TreeFiltered (Join-Path $RepoRoot 'static\assistive_harness') (Join-Path $stage 'static\assistive_harness') +New-Item -ItemType Directory -Force -Path (Join-Path $stage 'static\omni') | Out-Null +Copy-Item -LiteralPath (Join-Path $RepoRoot 'static\omni\omni-app.js') -Destination (Join-Path $stage 'static\omni\omni-app.js') + +$docNames = @( + 'VOICE_SKILL_HARNESS_AUDIT.md', + 'VOICE_SKILL_HARNESS_DESIGN.md', + 'VOICE_SKILL_HARNESS_EXECUTION_REPORT.md', + 'VOICE_SKILL_HARNESS_GO_NO_GO.md', + 'VOICE_SKILL_HARNESS_RERUN_COMMANDS.md', + 'VOICE_SKILL_HARNESS_CHANGED_FILES.md' +) +$docTarget = Join-Path $stage '_codex_context' +New-Item -ItemType Directory -Force -Path $docTarget | Out-Null +foreach ($name in $docNames) { + $text = Get-Content -Raw -Encoding UTF8 (Join-Path $RepoRoot "_codex_context\$name") + $text = $text.Replace($env:USERPROFILE, '%USERPROFILE%') + Set-Content -Encoding UTF8 -NoNewline -Path (Join-Path $docTarget $name) -Value $text +} + +$manifest = [ordered]@{ + project = 'Voice-Controlled Skill Harness' + phase = 'A_MINICPMO_BROWSER' + date = $Date + ready_for_phase_b = $false + feature_default = 'off' + excluded = @('model weights', 'raw audio/video', 'runs', 'credentials', 'absolute user paths') +} +$manifest | ConvertTo-Json -Depth 5 | Set-Content -Encoding UTF8 (Join-Path $stage 'MANIFEST.json') + +Compress-Archive -Path (Join-Path $stage '*') -DestinationPath $zipPath -CompressionLevel Optimal +$hash = (Get-FileHash -Algorithm SHA256 -LiteralPath $zipPath).Hash +[pscustomobject]@{Zip=$zipPath; SHA256=$hash} diff --git a/extensions/assistive_harness/phase_b/README.md b/extensions/assistive_harness/phase_b/README.md new file mode 100644 index 0000000..b7ffe10 --- /dev/null +++ b/extensions/assistive_harness/phase_b/README.md @@ -0,0 +1,114 @@ +# Rokid Phase B Adapter + +This adapter replaces the Phase A browser transport while preserving the +frozen Harness policy and the existing MiniCPM backend: + +```text +Rokid APK --JPEG/PCM over Wi-Fi--> Phase B Adapter :18080 + |-- audio.mirror/frame.shadow --> Harness :8021 + |-- audio_chunk + JPEG --------> Gateway :8040 + `-- MiniCPM audio -------------> PC speaker +``` + +The backend remains `llama.cpp-omni -> Worker -> Gateway :8040 -> MiniCPM-o +4.5`. The old ESP32 and Rokid bridge scripts remain reference implementations; +`demo_rokid_phase_b_harness.py` is the Phase B entry point. + +## Frozen control contract + +- `STOP`: block and flush PC playback immediately. Device PCM continues to + reach Harness ASR and Gateway with `force_listen=true`. +- `RESUME`: release the playback gate without replacing the Session. +- `RESET`: stop the old Gateway Session with light cleanup, increment the + generation, create a new Session using the idle prompt, and reject stale + output. After `restart_complete`, the PC plays a short two-note ready cue; + initial startup stays silent. Use `--no-session-ready-chime` to disable it, + or `--session-ready-chime-volume` (default `0.32`) to tune it. +- Skill activation, cancellation, and return-to-chat use the same replacement + flow with the prompt and slots supplied by Harness. +- After a Skill Session reaches `restart_complete`, Harness injects the + original one-shot task into that new Session. This makes read-text and + experimental obstacle activation answer once instead of only changing the + system prompt. +- Rokid PCM is repacketized into 100 ms float32 `audio.mirror` frames, matching + Phase A. A bounded drop-oldest queue prevents device capture from being + blocked by Gateway latency. +- PC speaker queue depth plus `--playback-echo-tail-s` (default `0.80`) keeps + EchoGuard active through buffered playback and its short acoustic tail. + +## Prerequisites + +Run the existing Worker/Gateway on port 8040 and the Assistive Harness on port +8021. Install optional adapter dependencies from `requirements-phase-b.txt` in +the same Python environment. + +The current Rokid APK must target the host computer on port 18080 and provide: + +- `POST /rokid/image` with raw JPEG bytes; +- `WS /rokid/audio` with 16 kHz mono signed PCM16 little-endian packets. + +## Start + +From the repository root: + +```powershell +python .\demo_rokid_phase_b_harness.py ` + --gateway localhost:8040 ` + --harness-url ws://127.0.0.1:8021/ws/control ` + --input-gain 12 ` + --image-rotate-cw 270 ` + --session-ready-chime-volume 0.32 ` + --playback-echo-tail-s 0.80 +``` + +The local Gateway currently uses plain `ws://`. Use `--gateway-tls` only when +the deployed Gateway endpoint actually serves `wss://`. First-round model audio +plays on the computer; `--no-play` disables playback for diagnostics. + +The Rokid SDK PCM observed on the current device is materially quieter than the +browser microphone signal. `--input-gain 12` is the initial device calibration; +gain is applied before both Harness and Gateway and saturates instead of wrapping +PCM16. Recalibrate from real-device RMS logs rather than lowering the shared +Phase A VAD threshold. + +The ready cue briefly suppresses Rokid PCM forwarding for the cue plus a short +guard interval so the laptop speaker notification is not immediately recycled +into Harness or MiniCPM-o. + +Streaming model text is aggregated by Session, generation and end-of-turn. It +is printed as `[AssistiveHarness][MODEL]` and written under the current run: + +- `model_events.jsonl` for structured records; +- `model_transcript.txt` for quick human review. + +This is generated model text, not ASR over the final TTS waveform. Keep a +session recording when exact spoken audio must be audited. + +Health and the normalized latest frame are available at: + +- `GET http://127.0.0.1:18080/health` +- `GET http://127.0.0.1:18080/capture` + +Healthy input is approximately 25 audio packets/s and 1 JPEG/s, with one audio +client, zero sustained queue drops, `harness_connected=true`, and +`gateway_status=running`. `device_input_ready=true` confirms that at least one +PCM packet or JPEG has reached this process. If `[ROKID][NO_INPUT]` appears, +restart Sensor Mode on the glasses (`STOP -> RUN`) after the PC listener is up; +an APK WebSocket disconnected by a PC restart may not reconnect by itself. + +Stop the adapter with `Ctrl+C` (or `Ctrl+Break` on Windows). The HTTP listener, +Rokid WebSocket handler, Gateway Session, Harness connection, and PC speaker are +closed with bounded waits. A 10-second process watchdog is the final fallback, +so a native audio cleanup stall cannot leave port 18080 occupied indefinitely. + +## First device acceptance order + +1. Ordinary speech produces a MiniCPM response on the PC speaker. +2. While it speaks, say `鍋滀竴涓媊; playback must stop and Harness must ACK + `stop_speech`. +3. Say `鎭㈠瀵硅瘽`; the same Session resumes. +4. Say `閲嶆柊寮濮媊; the Session ID and generation must change. +5. Activate `鎵剧墿`, `璇嗗瓧`, `鍦烘櫙鎻忚堪`, and supervised experimental `閬块殰`; + each switch must complete with a new Session and no old-generation output. + +Experimental obstacle output is not a navigation or safety guarantee. diff --git a/extensions/assistive_harness/phase_b/__init__.py b/extensions/assistive_harness/phase_b/__init__.py new file mode 100644 index 0000000..80d861b --- /dev/null +++ b/extensions/assistive_harness/phase_b/__init__.py @@ -0,0 +1,5 @@ +"""Device adapters for the Phase B glasses runtime.""" + +from .rokid_runtime import PhaseBRokidRuntime, RokidRuntimeConfig + +__all__ = ["PhaseBRokidRuntime", "RokidRuntimeConfig"] diff --git a/extensions/assistive_harness/phase_b/assets/reject_wav/aimed_wrong.wav b/extensions/assistive_harness/phase_b/assets/reject_wav/aimed_wrong.wav new file mode 100644 index 0000000..4fff8e2 Binary files /dev/null and b/extensions/assistive_harness/phase_b/assets/reject_wav/aimed_wrong.wav differ diff --git a/extensions/assistive_harness/phase_b/assets/reject_wav/manifest.json b/extensions/assistive_harness/phase_b/assets/reject_wav/manifest.json new file mode 100644 index 0000000..d7cb7b0 --- /dev/null +++ b/extensions/assistive_harness/phase_b/assets/reject_wav/manifest.json @@ -0,0 +1,92 @@ +{ + "no_frames": { + "text": "娌℃湁鎷垮埌鐢婚潰锛岃绋嶇瓑", + "wav": "no_frames.wav", + "strict_pass": true, + "text_match": true, + "sr_ok": true, + "dur_ok": true, + "text_out": "娌℃湁鎷垮埌鐢婚潰锛岃绋嶇瓑", + "dur": 2.2 + }, + "severe_shake": { + "text": "鐢婚潰鏅冨緱鍘夊锛岃淇濇寔涓嶅姩", + "wav": "severe_shake.wav", + "strict_pass": true, + "text_match": true, + "sr_ok": true, + "dur_ok": true, + "text_out": "鐢婚潰鏅冨緱鍘夊锛岃淇濇寔涓嶅姩", + "dur": 2.96 + }, + "unstable": { + "text": "鐢婚潰杩樺湪鏅冨姩锛岃鎷跨ǔ", + "wav": "unstable.wav", + "strict_pass": true, + "text_match": true, + "sr_ok": true, + "dur_ok": true, + "text_out": "鐢婚潰杩樺湪鏅冨姩锛岃鎷跨ǔ", + "dur": 3.04 + }, + "too_dark": { + "text": "鍏夌嚎澶殫锛岃鍒颁寒涓鐐圭殑鍦版柟", + "wav": "too_dark.wav", + "strict_pass": true, + "text_match": true, + "sr_ok": true, + "dur_ok": true, + "text_out": "鍏夌嚎澶殫锛岃鍒颁寒涓鐐圭殑鍦版柟", + "dur": 2.76 + }, + "aimed_wrong": { + "text": "濂藉儚娌″鍑嗭紝璇峰鍑嗙洰鏍", + "wav": "aimed_wrong.wav", + "strict_pass": true, + "text_match": true, + "sr_ok": true, + "dur_ok": true, + "text_out": "濂藉儚娌″鍑嗭紝璇峰鍑嗙洰鏍", + "dur": 2.96 + }, + "need_focus": { + "text": "瀵圭劍涓紝璇锋嬁绋充竴涓", + "wav": "need_focus.wav", + "strict_pass": true, + "text_match": true, + "sr_ok": true, + "dur_ok": true, + "text_out": "瀵圭劍涓紝璇锋嬁绋充竴涓", + "dur": 2.36 + }, + "orient": { + "text": "鐢婚潰濂藉儚鍙嶄簡锛岃鍊掕繃鏉", + "wav": "orient.wav", + "strict_pass": true, + "text_match": true, + "sr_ok": true, + "dur_ok": true, + "text_out": "鐢婚潰濂藉儚鍙嶄簡锛岃鍊掕繃鏉", + "dur": 3.4 + }, + "orient_flipped": { + "text": "鐢婚潰濂藉儚鍙嶄簡锛岃鍊掕繃鏉", + "wav": "orient_flipped.wav", + "strict_pass": true, + "text_match": true, + "sr_ok": true, + "dur_ok": true, + "text_out": "鐢婚潰濂藉儚鍙嶄簡锛岃鍊掕繃鏉", + "dur": 3.4 + }, + "orient_sideways": { + "text": "鐢婚潰濂藉儚姝簡锛岃鎽嗘", + "wav": "orient_sideways.wav", + "strict_pass": true, + "text_match": true, + "sr_ok": true, + "dur_ok": true, + "text_out": "鐢婚潰濂藉儚姝簡锛岃鎽嗘", + "dur": 2.44 + } +} \ No newline at end of file diff --git a/extensions/assistive_harness/phase_b/assets/reject_wav/need_focus.wav b/extensions/assistive_harness/phase_b/assets/reject_wav/need_focus.wav new file mode 100644 index 0000000..7b1f769 Binary files /dev/null and b/extensions/assistive_harness/phase_b/assets/reject_wav/need_focus.wav differ diff --git a/extensions/assistive_harness/phase_b/assets/reject_wav/no_frames.wav b/extensions/assistive_harness/phase_b/assets/reject_wav/no_frames.wav new file mode 100644 index 0000000..86806fc Binary files /dev/null and b/extensions/assistive_harness/phase_b/assets/reject_wav/no_frames.wav differ diff --git a/extensions/assistive_harness/phase_b/assets/reject_wav/orient.wav b/extensions/assistive_harness/phase_b/assets/reject_wav/orient.wav new file mode 100644 index 0000000..2b61d87 Binary files /dev/null and b/extensions/assistive_harness/phase_b/assets/reject_wav/orient.wav differ diff --git a/extensions/assistive_harness/phase_b/assets/reject_wav/orient_flipped.wav b/extensions/assistive_harness/phase_b/assets/reject_wav/orient_flipped.wav new file mode 100644 index 0000000..2b61d87 Binary files /dev/null and b/extensions/assistive_harness/phase_b/assets/reject_wav/orient_flipped.wav differ diff --git a/extensions/assistive_harness/phase_b/assets/reject_wav/orient_sideways.wav b/extensions/assistive_harness/phase_b/assets/reject_wav/orient_sideways.wav new file mode 100644 index 0000000..4d4f0b0 Binary files /dev/null and b/extensions/assistive_harness/phase_b/assets/reject_wav/orient_sideways.wav differ diff --git a/extensions/assistive_harness/phase_b/assets/reject_wav/severe_shake.wav b/extensions/assistive_harness/phase_b/assets/reject_wav/severe_shake.wav new file mode 100644 index 0000000..96247e5 Binary files /dev/null and b/extensions/assistive_harness/phase_b/assets/reject_wav/severe_shake.wav differ diff --git a/extensions/assistive_harness/phase_b/assets/reject_wav/too_dark.wav b/extensions/assistive_harness/phase_b/assets/reject_wav/too_dark.wav new file mode 100644 index 0000000..41d3f03 Binary files /dev/null and b/extensions/assistive_harness/phase_b/assets/reject_wav/too_dark.wav differ diff --git a/extensions/assistive_harness/phase_b/assets/reject_wav/unstable.wav b/extensions/assistive_harness/phase_b/assets/reject_wav/unstable.wav new file mode 100644 index 0000000..be9ac3f Binary files /dev/null and b/extensions/assistive_harness/phase_b/assets/reject_wav/unstable.wav differ diff --git a/extensions/assistive_harness/phase_b/bridge_ui.py b/extensions/assistive_harness/phase_b/bridge_ui.py new file mode 100644 index 0000000..a65ed15 --- /dev/null +++ b/extensions/assistive_harness/phase_b/bridge_ui.py @@ -0,0 +1,362 @@ +"""bridge_ui.py 鈥 ESP32 duplex bridge 鐨 Web 瑙傛祴/鍥炴斁鏈嶅姟 + +URL: + / 瀹炴椂瑙傛祴(鎽勫儚澶 + 瀛楀箷 + 鐘舵) + /replay session 鍒楄〃 + /replay/ session 鍥炴斁(鐩存帴鎾 live_session.mp4,瀛楀箷璺熼殢) + +渚濊禆 recorder_live.py v5+ 鍐欏埌 session_dir 鐨勬垚鍝: + live_session.mp4 鈫 鎾斁涓讳綋(ffmpeg 宸插榻) + live_session.m4a 鈫 娌℃湁瑙嗛甯ф椂鐨勭函闊抽鍥為 + live_user.wav 鈫 璇婃柇涓嬭浇 + live_ai.wav 鈫 璇婃柇涓嬭浇 + events.jsonl 鈫 瀛楀箷/浜嬩欢婧 + meta.json 鈫 浼氳瘽鍏冧俊鎭 +""" + +from __future__ import annotations +from typing import Optional, Set, Callable +import json +import logging +from pathlib import Path +from typing import Optional, Set + +from aiohttp import web + +LOGGER = logging.getLogger("bridge_ui") + +# 妯℃澘鐩綍(鍜 bridge_ui.py 鍚岀骇鐨 templates/) +TEMPLATES_DIR = Path(__file__).parent / "templates" + +# 寮鍙戞湡鎯"鏀瑰畬鍒锋柊娴忚鍣ㄥ氨鐢熸晥"灏辫 True;鐢熶骇鐜璁 False 鍙涓娆 +HOT_RELOAD_TEMPLATES = True + +_template_cache: dict[str, str] = {} + +def _load_template(name: str) -> str: + """璇诲彇 templates/銆侶OT_RELOAD_TEMPLATES=True 鏃舵瘡娆¢兘閲嶈銆""" + if not HOT_RELOAD_TEMPLATES and name in _template_cache: + return _template_cache[name] + path = TEMPLATES_DIR / name + try: + text = path.read_text(encoding="utf-8") + except FileNotFoundError: + LOGGER.error("[UI] template not found: %s", path) + return f"

Template not found: {name}

" + _template_cache[name] = text + return text + + +# ============================================================ +# Server +# ============================================================ + +class WebUIServer: + # 榛樿鍙粦鍥炵幆锛氳繖涓湇鍔℃妸绗竴瑙嗚鐢婚潰銆乻ession 鍏冩暟鎹佸師濮 user/AI wav + # 鍏ㄩ儴鏃犺璇佸湴鏆撮湶鍑哄幓銆傜粦 0.0.0.0 绛変簬鎶婂畠浠斁缁欐暣涓眬鍩熺綉锛 + # 瑕侀偅鏍峰繀椤荤敱璋冪敤鏂规樉寮忔寚瀹 host銆 + def __init__(self, port: int = 8080, host: str = "127.0.0.1", + sessions_root: Path = Path("./sessions"), + stop_callback: Optional[Callable[[], None]] = None, + mode_info: Optional[dict] = None): + self.port = port + self.host = host + self.sessions_root = Path(sessions_root) + self.live_clients: Set[web.WebSocketResponse] = set() + self._runner: Optional[web.AppRunner] = None + self._stop_callback = stop_callback # 鈫 鏂板 + self._stop_fired = False # 鈫 鏂板,闃查噸鍏 + self.mode_info = mode_info or {"mode": "live"} + + async def start(self) -> None: + app = web.Application() + app.router.add_get('/', self._h_live) + app.router.add_get('/replay', self._h_replay_index) + app.router.add_get('/replay/{sid}', self._h_replay) + app.router.add_get('/live_ws', self._h_live_ws) + app.router.add_get('/api/sessions', self._h_list) + app.router.add_get('/api/session/{sid}/events', self._h_events) + app.router.add_get('/api/session/{sid}/meta', self._h_meta) + # 鈥斺 v2: 鐩存帴鍚愭垚鍝佹枃浠,FileResponse 鑷甫 HTTP Range 鏀寔 鈥斺 + app.router.add_get('/api/session/{sid}/video', self._h_video) + app.router.add_get('/api/session/{sid}/audio', self._h_audio_only) + app.router.add_get('/api/session/{sid}/user_audio.wav', self._h_user_wav) + app.router.add_get('/api/session/{sid}/ai_audio.wav', self._h_ai_wav) + # 鍏煎鑰佽矾鐢(鍙兘鏃х増 events.jsonl 寮曠敤 images/xxx.jpg) + app.router.add_get('/api/session/{sid}/{path:images/.+}', self._h_image_legacy) + app.router.add_get('/api/session/{sid}/{path:live_images/.+}', self._h_image_live) + app.router.add_post('/api/stop', self._h_stop) + self._runner = web.AppRunner(app) + await self._runner.setup() + # 鍥炵幆榛樿鍚屾椂缁 IPv4 鍜 IPv6銆 + # 鍙粦 "127.0.0.1" 鐨勮瘽鏄**绾 IPv4**锛岃 Windows 涓婃祻瑙堝櫒鎶 + # `localhost` 浼樺厛瑙f瀽鎴 ::1 鐨勬儏鍐靛緢甯歌 鈥斺 閭f椂 panel 鍐呭祵鐨 + # http://localhost:8080 浼氳繛涓嶄笂锛堣〃鐜板氨鏄涓瑙嗚涓鐗囩┖鐧斤級銆 + # 缁戞垚 ["127.0.0.1", "::1"] 涓よ竟閮借鐩栵紝鍙堜笉鏆撮湶鍒板眬鍩熺綉銆 + _hosts = self.host + if self.host in ("127.0.0.1", "localhost", "::1"): + _hosts = ["127.0.0.1", "::1"] + try: + site = web.TCPSite(self._runner, _hosts, self.port) + await site.start() + except OSError as e: + # 绯荤粺娌″紑 IPv6 涔嬬被锛氶鍥炲崟鍦板潃锛屽埆璁╂暣涓涓瑙嗚璧蜂笉鏉 + LOGGER.warning("[UI] 缁戝畾 %s 澶辫触(%s)锛岄鍥 %s", _hosts, e, self.host) + site = web.TCPSite(self._runner, self.host, self.port) + await site.start() + LOGGER.info("[UI] http://localhost:%d (live + replay) bind=%s", + self.port, _hosts) + if self.host not in ("127.0.0.1", "localhost", "::1"): + LOGGER.warning( + "[UI] !! 缁戝畾鍦 %s 鈥斺 绗竴瑙嗚鐢婚潰銆乻ession 鍏冩暟鎹" + "鍘熷 user/AI 褰曢煶灏嗘棤璁よ瘉鍦版毚闇茬粰灞鍩熺綉鍐呬换浣曚汉銆", self.host) + + async def stop(self) -> None: + for ws in list(self.live_clients): + try: await ws.close() + except Exception: pass + if self._runner: + await self._runner.cleanup() + + async def emit(self, evt: dict) -> None: + if not self.live_clients: + return + msg = json.dumps(evt, ensure_ascii=False) + dead = [] + for ws in list(self.live_clients): + try: await ws.send_str(msg) + except Exception: dead.append(ws) + for ws in dead: + self.live_clients.discard(ws) + + # ---- handlers ---- + async def _h_live(self, request): + return web.Response(text=_load_template("live.html"), content_type='text/html') + + async def _h_stop(self, request): + if self._stop_fired: + return web.json_response({"ok": True, "already": True}) + self._stop_fired = True + LOGGER.info("[UI] /api/stop triggered by browser") + # 閫氱煡鎵鏈 live 瀹㈡埛绔"瑕佸叧浜",鍓嶇鍙互鎶婃寜閽敼鎴"Saving..." + try: + await self.emit({"type": "stopping"}) + except Exception: + pass + if self._stop_callback is not None: + try: + self._stop_callback() + except Exception as e: + LOGGER.warning("[UI] stop_callback err: %s", e) + return web.json_response({"ok": True}) + + async def _h_replay_index(self, request): + return web.Response(text=_load_template("replay_index.html"), content_type='text/html') + + async def _h_replay(self, request): + return web.Response(text=_load_template("replay.html"), content_type='text/html') + + async def _h_live_ws(self, request): + ws = web.WebSocketResponse(heartbeat=30) + await ws.prepare(request) + self.live_clients.add(ws) + + LOGGER.info("[UI] live client +1 (total=%d)", len(self.live_clients)) + try: + await ws.send_str(json.dumps( + {"type": "mode", **self.mode_info}, ensure_ascii=False)) + except Exception: + pass + + try: + async for _ in ws: pass + finally: + self.live_clients.discard(ws) + LOGGER.info("[UI] live client -1 (total=%d)", len(self.live_clients)) + return ws + + async def _h_list(self, request): + sessions = [] + if self.sessions_root.exists(): + for d in sorted(self.sessions_root.iterdir(), reverse=True): + if not d.is_dir(): + continue + meta_p = d / "meta.json" + meta = {} + if meta_p.exists(): + try: + meta = json.loads(meta_p.read_text(encoding='utf-8')) + except Exception: + pass + if (d / "live_session.mp4").exists(): + media = "mp4" + elif (d / "live_session.m4a").exists(): + media = "m4a" + else: + media = None + sessions.append({ + "id": d.name, + "tag": meta.get("session_tag", d.name), + "start_time": meta.get("start_time"), + "end_time": meta.get("end_time"), + "stats": meta.get("stats", {}), + "media": media, + }) + return web.json_response({"sessions": sessions}) + + def _safe(self, sid: str) -> Optional[Path]: + if "/" in sid or ".." in sid: + return None + p = (self.sessions_root / sid).resolve() + try: + p.relative_to(self.sessions_root.resolve()) + except ValueError: + return None + if not p.is_dir(): + return None + return p + + async def _h_meta(self, request): + p = self._safe(request.match_info['sid']) + if not p: + return web.json_response({"error": "not found"}, status=404) + meta = p / "meta.json" + if not meta.exists(): + return web.json_response({"error": "no meta"}, status=404) + return web.json_response(json.loads(meta.read_text(encoding='utf-8'))) + + async def _h_events(self, request): + p = self._safe(request.match_info['sid']) + if not p: + return web.json_response({"error": "not found"}, status=404) + ev = p / "events.jsonl" + if not ev.exists(): + return web.json_response({"events": []}) + out = [] + for line in ev.read_text(encoding='utf-8').splitlines(): + line = line.strip() + if not line: + continue + try: + out.append(json.loads(line)) + except Exception: + pass + return web.json_response({"events": out}) + + # ---- 濯掍綋:鐩存帴 FileResponse(鏀寔 Range,鍙嫋鍔ㄨ繘搴︽潯)---- + + async def _h_video(self, request): + p = self._safe(request.match_info['sid']) + if not p: + return web.Response(status=404) + mp4 = p / "live_session.mp4" + if not mp4.exists(): + return web.Response(status=404, text="no live_session.mp4") + return web.FileResponse(mp4, headers={"Content-Type": "video/mp4"}) + + async def _h_audio_only(self, request): + p = self._safe(request.match_info['sid']) + if not p: + return web.Response(status=404) + m4a = p / "live_session.m4a" + if not m4a.exists(): + return web.Response(status=404, text="no live_session.m4a") + return web.FileResponse(m4a, headers={"Content-Type": "audio/mp4"}) + + async def _h_user_wav(self, request): + p = self._safe(request.match_info['sid']) + if not p: + return web.Response(status=404) + w = p / "live_user.wav" + if not w.exists(): + return web.Response(status=404) + return web.FileResponse(w, headers={"Content-Type": "audio/wav"}) + + async def _h_ai_wav(self, request): + p = self._safe(request.match_info['sid']) + if not p: + return web.Response(status=404) + w = p / "live_ai.wav" + if not w.exists(): + return web.Response(status=404) + return web.FileResponse(w, headers={"Content-Type": "audio/wav"}) + + async def _h_image_live(self, request): + return await self._serve_image(request, subdir="live_images") + + async def _h_image_legacy(self, request): + return await self._serve_image(request, subdir="images") + + async def _serve_image(self, request, subdir: str): + p = self._safe(request.match_info['sid']) + if not p: + return web.Response(status=404) + rel = request.match_info['path'] + # rel 褰㈠ "live_images/img_00001.jpg" 鎴 "images/xxx.jpg" + # 杩欓噷鍙彇鏈熬鏂囦欢鍚,闃茬┛瓒 + name = Path(rel).name + if "/" in name or ".." in name or not name: + return web.Response(status=400) + img = p / subdir / name + if not img.exists(): + return web.Response(status=404) + return web.FileResponse(img, headers={"Content-Type": "image/jpeg"}) + + +# ============================================================ +# Standalone entry +# ============================================================ + +def _main() -> None: + import argparse + import asyncio + + parser = argparse.ArgumentParser( + description="Bridge UI 鈥 standalone replay server (no ESP32 / no model)" + ) + parser.add_argument("--port", type=int, default=8080) + parser.add_argument("--host", default="127.0.0.1", + help="缁戝畾鍦板潃銆傞粯璁ゅ彧缁戝洖鐜 鈥斺 鏈湇鍔℃棤璁よ瘉鍦版彁渚" + "绗竴瑙嗚鐢婚潰銆乻ession 鍏冩暟鎹拰鍘熷 user/AI wav锛" + "濉 0.0.0.0 浼氭妸杩欎簺鏆撮湶缁欐暣涓眬鍩熺綉") + parser.add_argument("--sessions", default="./sessions", + help="sessions 鏍圭洰褰(榛樿 ./sessions)") + args = parser.parse_args() + + logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(name)s: %(message)s", + ) + + sessions_root = Path(args.sessions).resolve() + if not sessions_root.exists(): + LOGGER.warning("[UI] sessions dir 涓嶅瓨鍦: %s(涔嬪悗褰曞埌杩欓噷灏辫兘鐪嬭)", + sessions_root) + + server = WebUIServer( + port=args.port, + host=args.host, + sessions_root=sessions_root + ) + + async def _run(): + await server.start() + LOGGER.info("[UI] standalone mode 鈥 replay only") + LOGGER.info("[UI] sessions root: %s", sessions_root) + LOGGER.info("[UI] open http://localhost:%d/replay", args.port) + try: + while True: + await asyncio.sleep(3600) + except (KeyboardInterrupt, asyncio.CancelledError): + pass + finally: + await server.stop() + + try: + asyncio.run(_run()) + except KeyboardInterrupt: + print("\n[UI] bye.") + + +if __name__ == "__main__": + _main() \ No newline at end of file diff --git a/extensions/assistive_harness/phase_b/cam_pipeline_v2.py b/extensions/assistive_harness/phase_b/cam_pipeline_v2.py new file mode 100644 index 0000000..0659a48 --- /dev/null +++ b/extensions/assistive_harness/phase_b/cam_pipeline_v2.py @@ -0,0 +1,2298 @@ +# -*- coding: utf-8 -*- +"""cam_tuner.py 鈥 ESP32 鐩告満璋冨弬鍙 (褰㈡1: 绾皟鍙, 涓嶆帴妯″瀷) + +鐩殑: 涓杈圭湅 TCP live 鐢婚潰, 涓杈圭儹璋冨垎杈ㄧ巼/瀵圭劍/quality, 瀹炴椂鐪嬫瘡甯х殑 + 鎷夋櫘鎷夋柉娓呮櫚搴﹀垎 + 浜害 鈥斺 涓轰笁绾ф紡鏂椾竴绾ф爣瀹氶槇鍊笺 + +鐢ㄦ硶: + python cam_tuner.py --ip 10.100.7.68 + 娴忚鍣ㄥ紑 http://localhost:8080 + +娉ㄦ剰 (纭害鏉): + 鍥轰欢 TCP 5000 鏄崟瀹㈡埛绔覆琛屻傛湰宸ュ叿鐙崰 TCP 鈥斺 璋冨弬鏃朵笉瑕佸悓鏃惰窇涓荤▼搴 + (pc_vlm / demo_esp32), 鍚﹀垯鎶㈠悓涓鏉 TCP 杩炴帴浼氶敊浣嶃 + +渚濊禆: aiohttp, opencv-python(cv2), numpy, requests +""" + +from __future__ import annotations + +import argparse +import asyncio +import socket +import struct +import threading +import time +from pathlib import Path +from typing import Optional + +import cv2 +import numpy as np +import requests +from aiohttp import web + + +# ============================================================ +# TCP 5000 鎶撳抚 (澶嶇敤宸查獙璇佸崗璁; 鍗曡繛鎺 + 閿, 鍚庡彴绾跨▼鎸佹湁) +# ============================================================ +class TCPImageClient: + def __init__(self, ip: str, port: int = 5000, timeout: float = 2.0): + self.ip, self.port, self.timeout = ip, port, timeout + self._sock = None + self._lock = threading.Lock() + + def _connect_locked(self) -> bool: + try: + self._sock = socket.create_connection((self.ip, self.port), timeout=self.timeout) + self._sock.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1) + self._sock.settimeout(self.timeout) + return True + except Exception: + self._sock = None + return False + + def _recv_exactly(self, n: int) -> Optional[bytes]: + buf = b"" + while len(buf) < n: + chunk = self._sock.recv(n - len(buf)) + if not chunk: + return None + buf += chunk + return buf + + def capture(self) -> Optional[bytes]: + with self._lock: + for attempt in (1, 2): + if self._sock is None and not self._connect_locked(): + return None + try: + self._sock.sendall(b"\x01") + hdr = self._recv_exactly(20) + if not hdr: + raise ConnectionError("no header") + magic = struct.unpack_from(" bool: + try: + r = self.sess.get(f"{self.base}/control", + params={"var": var, "val": val}, timeout=self.timeout) + return r.status_code == 200 + except Exception: + return False + + def set_resolution(self, name: str) -> bool: + v = FRAMESIZE.get(name.upper()) + return self.control("framesize", v) if v is not None else False + + def set_quality(self, q: int) -> bool: + return self.control("quality", max(0, min(63, int(q)))) + + def trigger_af(self) -> bool: + # 鍗曟瀵圭劍: 鐩存帴鍐 OV5640 瀵勫瓨鍣 0x3022=0x03 (single auto focus)銆 + # 鍥轰欢鐢ㄦ爣鍑 esp32-camera web server, 娌℃湁瀹炵幇 /control?var=af (af 闈炴爣鍑嗗彉閲), + # 涔嬪墠鍏堣瘯 control("af") 浼"鍋囨垚鍔"(鍥轰欢蹇界暐)鑰屼笉鐪熷鐒 鈥斺 鏁呯洿鎺ヨ蛋 /reg銆 + # 鍥轰欢绔簲璁 g_auto_af=false, 鍏虫帀姣忕瀹氭椂纭鐒, 瀵圭劍瀹屽叏鐢辨鎸夐渶瑙﹀彂銆 + try: + r = self.sess.get(f"{self.base}/reg", + params={"reg": 0x3022, "mask": 0xff, "val": 0x03}, + timeout=self.timeout) + return r.status_code == 200 + except Exception: + return False + + +def rotate_jpeg(jpg: bytes, deg: int) -> bytes: + """鍚庣鐪熸棆杞 (椤烘椂閽 deg鈭坽0,90,180,270})銆傝繑鍥炴棆杞悗鐨 JPEG bytes銆 + 鐪熻浆鑰岄潪鍙浆鏄剧ず: 鍥犱负鏈缁堣嚜鍔ㄦ棆杞瓥鐣ュ垽鏂悗浼氱湡鍠傜粰妯″瀷, 鏁版嵁璺緞瑕佷竴鑷; + 涓旀竻鏅板害鍒嗗熀浜庤浆鍚庡浘绠, 鍙獙璇'鏃嬭浆涓嶆敼鍙樻竻鏅板害'(閿愬埄搴︿笌鏈濆悜姝d氦)銆""" + d = deg % 360 + if d == 0: + return jpg + arr = np.frombuffer(jpg, dtype=np.uint8) + img = cv2.imdecode(arr, cv2.IMREAD_COLOR) + if img is None: + return jpg + if d == 90: + img = cv2.rotate(img, cv2.ROTATE_90_CLOCKWISE) + elif d == 180: + img = cv2.rotate(img, cv2.ROTATE_180) + elif d == 270: + img = cv2.rotate(img, cv2.ROTATE_90_COUNTERCLOCKWISE) + else: + return jpg + ok, enc = cv2.imencode(".jpg", img, [cv2.IMWRITE_JPEG_QUALITY, 90]) + return enc.tobytes() if ok else jpg + + +def _decode_gray(jpg: bytes): + arr = np.frombuffer(jpg, dtype=np.uint8) + return cv2.imdecode(arr, cv2.IMREAD_GRAYSCALE) + + +def frame_motion(jpg_a: bytes, jpg_b: bytes) -> float: + """涓ゅ抚鐏板害骞冲潎缁濆宸 (0-255)銆傚ぇ = 鐢婚潰鍦ㄥ姩 (蹇熺Щ鍔/杞ご)銆 + **鍏堥珮鏂ā绯婂啀绠楀樊**: 鎶规帀瀵嗛泦鏂囧瓧鐨勯珮棰戠粏鑺, 鍙繚鐣欏畯瑙備綅绉 鈥斺 + 鍚﹀垯瀵嗛泦鏂囧瓧闈(鎴愬垎闈)鎵嬫寔杞诲井绉诲姩鏃舵瘡涓皬瀛楄竟缂橀兘浜х敓澶ч噺鍍忕礌宸, motion 铏氶珮, + 鎶婂彲璇荤殑鍥捐鍒ゆ垚"鍦ㄥ姩"(鐪熸満+OCR鏍″噯璇佸疄鎴愬垎闈㈠彲璇诲嵈琚叏鎷)銆 + 婕忔枟浜岀骇鐢: 涓鎵瑰抚鐩搁偦 diff 閮藉ぇ -> 鐢婚潰涓嶇ǔ -> 鎷掔粷(蹇靛瓧蹇呴敊, 璁╃敤鎴峰仠绋)銆""" + a, b = _decode_gray(jpg_a), _decode_gray(jpg_b) + if a is None or b is None: + return 0.0 + if a.shape != b.shape: + b = cv2.resize(b, (a.shape[1], a.shape[0])) + a = cv2.GaussianBlur(a, (7, 7), 0) + b = cv2.GaussianBlur(b, (7, 7), 0) + return float(np.abs(a.astype(np.int16) - b.astype(np.int16)).mean()) + + +def optical_flow_motion(jpg_a: bytes, jpg_b: bytes, downscale: float = 0.35) -> float: + """涓ゅ抚绋犲瘑鍏夋祦(Farneback)鐨勫钩鍧囦綅绉诲箙搴(鍍忕礌)銆傜墿鐞嗗惈涔夋槑纭佸彲瑙i噴: + 鐢婚潰鏁翠綋绉诲姩浜嗗灏戝儚绱犮傜敤浜庛愪弗閲嶆檭鍔ㄤ竴鍒鍒囥戔斺 鐩告満澶у箙浣嶇Щ鏃跺厜娴佸箙搴﹀ぇ銆 + 姣 frame_motion(閫愬儚绱犲樊)鏇撮瞾妫: 鍏夋祦鐪嬭繍鍔ㄧ煝閲忓満, 涓嶈瀵嗛泦鏂囧瓧鐨勯珮棰戠粏鑺傚共鎵 + (瀵嗛泦瀛楄交寰Щ鍔 -> 鍏夋祦涓鑷村皬浣嶇Щ; 閫愬儚绱犲樊鍗磋櫄楂)銆傞檷閲囨牱鎻愰(鍒"涓ラ噸"澶熺敤)銆""" + a, b = _decode_gray(jpg_a), _decode_gray(jpg_b) + if a is None or b is None: + return 0.0 + if a.shape != b.shape: + b = cv2.resize(b, (a.shape[1], a.shape[0])) + if downscale != 1.0: + a = cv2.resize(a, None, fx=downscale, fy=downscale, interpolation=cv2.INTER_AREA) + b = cv2.resize(b, None, fx=downscale, fy=downscale, interpolation=cv2.INTER_AREA) + flow = cv2.calcOpticalFlowFarneback(a, b, None, + pyr_scale=0.5, levels=3, winsize=15, + iterations=3, poly_n=5, poly_sigma=1.2, flags=0) + mag = np.sqrt(flow[..., 0] ** 2 + flow[..., 1] ** 2) + # 浣嶇Щ骞呭害鎸夐檷閲囨牱姣斾緥杩樺師鍒板師鍥惧昂搴, 渚夸簬鐢ㄥ師鍥惧儚绱犲崟浣嶈闃堝 + return float(mag.mean() / max(downscale, 1e-6)) + + +def bg_signals(jpg_bytes): + """鍊欓'鑳屾櫙骞叉壈'淇″彿(閮藉彲瑙i噴, 妫娴嬬敾闈㈡湁棰濆鑳屾櫙 -> 鎻愮ず鎷胯繎/瀵瑰噯)銆 + 骞插噣鐧界焊鏍囩: 澶х墖鍧囧寑鐧 + 涓棿灏戦噺榛戝瓧; 鏈夎儗鏅(鐏/宸ヤ綅/澶╄姳鏉)鏃朵俊鍙峰紓甯搞 + 涔熷惈娓呮櫚搴︾被(local_sharp/worst_block/sharp_ratio)渚涜瘎鍒嗐傝繑鍥 dict銆""" + arr = np.frombuffer(jpg_bytes, np.uint8) + g = cv2.imdecode(arr, cv2.IMREAD_GRAYSCALE) + if g is None: + return {} + h, w = g.shape + # 淇″彿1: 浜害鐩存柟鍥剧喌 鈥斺 骞插噣鏍囩(鐧藉簳+榛戝瓧)鍒嗗竷闆嗕腑鐔典綆; 鑳屾櫙鏉->鐔甸珮 + hist = cv2.calcHist([g], [0], None, [32], [0, 256]).flatten() + p = hist / (hist.sum() + 1e-9) + entropy = float(-np.sum(p * np.log2(p + 1e-12))) + # 淇″彿2: 鏈澶у潎鍖(浣庢柟宸)鍖哄崰姣 鈥斺 鐧界焊鏈夊ぇ鐗囧潎鍖鐧; 鍗犳瘮浣=鐢婚潰鏉備贡 + blur = cv2.GaussianBlur(g, (5, 5), 0) + localvar = cv2.blur((g.astype(np.float32) - blur.astype(np.float32)) ** 2, (15, 15)) + lowvar = (localvar < 30).astype(np.uint8) + n, labels, stats, _ = cv2.connectedComponentsWithStats(lowvar) + max_uniform = float(stats[1:, cv2.CC_STAT_AREA].max() / g.size) if n > 1 else 0.0 + # 淇″彿3: 杈圭紭绌洪棿鍒嗗竷 鈥斺 骞插噣鏍囩杈圭紭闆嗕腑涓棿(鏂囧瓧); 鑳屾櫙浣胯竟缂樻暎甯/钀藉鍥 + edges = cv2.Canny(g, 50, 150) + ys, xs = np.where(edges > 0) + if len(xs) > 10: + cx, cy = w / 2, h / 2 + d = np.sqrt(((xs - cx) / w) ** 2 + ((ys - cy) / h) ** 2) + edge_spread = float(d.mean()) + edge_periph = float((d > 0.35).mean()) # 杈圭紭钀藉鍥(杩滅涓績)姣斾緥 + else: + edge_spread = 0.0; edge_periph = 0.0 + # 淇″彿4: local_sharp / worst_block 鈥斺 鏈夐珮瀵规瘮鑳屾櫙(鐏)鏃 local铏氶珮鑰屾暣浣撶硦 + lap = cv2.Laplacian(g, cv2.CV_64F) + local_sharp = float(lap.var()) + grid = 6 + vals = [] + for i in range(grid): + for j in range(grid): + blk = g[i*h//grid:(i+1)*h//grid, j*w//grid:(j+1)*w//grid] + if blk.size == 0: + continue + e = cv2.Canny(blk, 50, 150) + if (e > 0).mean() < 0.006: + continue + vals.append(float(cv2.Laplacian(blk, cv2.CV_64F).var())) + worst = min(vals) if vals else 0.0 + sharp_ratio = float(local_sharp / (worst + 1e-6)) if worst > 0 else 0.0 + return { + "bright_entropy": round(entropy, 2), + "max_uniform": round(max_uniform, 3), + "edge_spread": round(edge_spread, 3), + "edge_periph": round(edge_periph, 3), + "sharp_ratio": round(sharp_ratio, 1), + "local_sharp": round(local_sharp, 1), + "worst_block": round(worst, 1), + } + + +# ============================================================ +# 涓夌骇婕忔枟 (褰㈡2鏍稿績) +# 杈撳叆: 涓鎵 N 甯 (bytes list) +# 涓绾 鍗曞抚璐ㄩ噺绛: 澶硦/澶殫/绌虹櫧 鐨勫抚鏍囪涓嶅悎鏍 +# 浜岀骇 绋冲畾鎬у垽瀹: 鐩搁偦甯 motion 閮借繃澶 -> 鏁存壒鎷掔粷 (鐢婚潰涓嶇ǔ) +# 涓夌骇 閫変紭: 鍚堟牸甯ч噷閫 sharpness 鏈楂樼殑 = best +# 鍏ㄤ笉鍚堟牸 / 涓嶇ǔ -> 杩斿洖鎷掔粷 (瀹夊叏闂, 涓嶅杺妯″瀷) +# +# 闃堝艰鏄: 鍒嗚鲸鐜囧浐瀹 HD 鍚庨拡瀵 HD 鏍囧畾銆傝繖閲岀粰鐨勬槸鍒濆榛樿鍊, +# 浣犳埓鐪奸暅瀹炴祴鍚庢寜 CSV 鏁版嵁璋 (灏ゅ叾 SHARP_MIN / MOTION_MAX)銆 +# sharp 浣 + edge 浣 = 绌虹櫧(鎷掔粷, 璇"娌$湅鍒版枃瀛"); 鐢 EDGE_MIN 鍏溿 +# ============================================================ +import math + + +def _g_sat(x: float, x0: float) -> float: + """Michaelis-Menten 楗卞拰: x/(x+x0) -> [0,1)銆倄0=鍗婇ケ鍜岀偣銆""" + x = max(0.0, x) + return x / (x + x0) if (x + x0) > 0 else 0.0 + + +def _g_light(b: float, lo: float, hi: float, soft: float) -> float: + """浜害鍝嶅簲: 澶殫/杩囨洕闄嶅垎, 涓棿骞冲彴=1銆備袱 sigmoid 鐩稿す銆""" + def sig(z): + try: + return 1.0 / (1.0 + math.exp(-z)) + except OverflowError: + return 0.0 if z < 0 else 1.0 + return sig((b - lo) / soft) * (1 - sig((b - hi) / soft)) + + +def frame_score(metrics: dict, rel_change: float, cfg: "FunnelConfig") -> dict: + """瀵逛竴甯х畻 quality + focus_gain + 鍚勫垎閲 (渚涜В閲/璁板綍)銆 + rel_change: 璇ュ抚鐩稿涓婁竴甯х殑 sharp 鐩稿鍙樺寲 (绋冲畾鎬ц緭鍏); 棣栧抚浼 0銆""" + # 娓呮櫚搴︾敤 worst_block(鏈绯婄殑鏈夊唴瀹瑰潡) 鑰岄潪鍏ㄥ眬 sharpness 鈥斺 + # 鍏ㄥ眬鎷夋櫘鎷夋柉琚ぇ闈㈢Н鐧藉簳绋閲(鐧藉簳鏃犺竟缂, 鎷夋櫘鎷夋柉浣), 鎶"娓呮櫚鐨勭櫧鐩掑瓙鏍囩"璇垽鎴愮硦, + # 瑙﹀彂鍋囩殑 need_focus銆倃orst_block 鍙湅鏈夊唴瀹圭殑鏈绯婂潡, 涓嶈鐧藉簳鎷変綆銆 + # 鐪熸満+OCR鏍″噯: 涓嶅彲璇诲抚 worst_block<4.6, 鍙>5.5 -> 鍒嗙晫~5, 鏁 x0 鐢 worst_x0(~5閲忕骇)銆 + worst = metrics.get("worst_block", None) + if worst is None: + worst = metrics.get("sharpness", 0.0) # 鍏滃簳: 鏃 worst_block 鏃堕鍥炲叏灞 + g_sharp = _g_sat(worst, cfg.worst_x0) + g_stable = max(0.0, 1.0 - min(1.0, rel_change)) # rel_change 宸插綊涓鍖栧埌[0,1] + g_content = _g_sat(metrics.get("edge_density", 0.0), cfg.edge_x0) + g_light = _g_light(metrics.get("brightness", 0.0), cfg.light_lo, cfg.light_hi, cfg.light_soft) + + wsum = cfg.w_sharp + cfg.w_stable + cfg.w_content + cfg.w_light + base = (cfg.w_sharp * g_sharp + cfg.w_stable * g_stable + + cfg.w_content * g_content + cfg.w_light * g_light) / wsum + sharp_gate = g_sharp ** cfg.sharp_gate_gamma # 蹇靛瓧纭姹: 绯婂垯鍘嬪垎 + quality = base * sharp_gate + # focus_gain: 绋 + 绯 鏃惰瀵圭劍銆傚 content 鐢"娓╁拰渚濊禆"(sqrt)鑰岄潪纭箻 鈥斺 + # 閲嶅害澶辩劍浼氭妸杈圭紭涔熺硦娌(g_content 浣), 浣嗗畠鎭版伆鏈璇ュ鐒, 涓嶈兘鍥犳褰掗浂; + # sqrt 璁╀綆鍐呭鏃 focus_gain 闄嶄綆浣嗕繚鐣欒Е鍙戞満浼氥傜湡绌虹櫧鐢 g_content 鏋佷綆 + + # g_stable 楂樼殑缁勫悎鍙﹁璇嗗埆(瑙 run_funnel 鎷掔粷褰掑洜)銆 + focus_gain = g_stable * (1 - g_sharp) * math.sqrt(max(0.0, g_content)) + return { + "quality": round(quality, 3), + "focus_gain": round(focus_gain, 3), + "g_sharp": round(g_sharp, 3), + "g_stable": round(g_stable, 3), + "g_content": round(g_content, 3), + "g_light": round(g_light, 3), + } + + +class FunnelConfig: + """璇勫垎鍑芥暟 f 鐨勫弬鏁 (鏇夸唬纭槇鍊 if)銆傛瘡甯х畻涓涓繛缁 quality 鍒 + focus_gain, + 鍐崇瓥鍩轰簬鍒嗘暟鑰岄潪绂绘暎闃堝 鈥斺 鍙В閲娿佸彲 ablation銆佸彲鐢ㄦ暟鎹牎鍑嗐 + + quality = [危 wi路gi] 路 g_sharp^gamma (sharp 浣滀箻鎬ч棬, 蹇靛瓧纭姹) + g_sharp = sharp/(sharp+sharp_x0) 娓呮櫚搴﹂ケ鍜屽搷搴 + g_stable = 1 - min(1, rel_change/change_ref) 甯ч棿鐩稿鍙樺寲瓒婂皬瓒婄ǔ + g_content = edge/(edge+edge_x0) 鏈夋棤鍐呭(鍖哄垎绯/绌) + g_light = 浜害鍙宻igmoid(鏆/杩囨洕閮介檷) + focus_gain = g_stable路(1-g_sharp)路g_content 绋+绯+鏈夊唴瀹 -> 璇ュ鐒 + 鍐崇瓥: max(quality) >= ACCEPT_Q -> 閫夎甯; 鍚﹀垯鐪 max(focus_gain) >= FOCUS_G + -> 瑙﹀彂瀵圭劍鍐嶉噰; 閮戒綆 -> 鎷掔粷銆 + 鍙傛暟鏄墿鐞嗗姩鏈哄垵鍊, 鎴寸溂闀滈噰鏁版嵁鍚庡彲鏍″噯/鎷熷悎 (鏂规硶 vs 璋冨弬鐨勫尯鍒)銆 + """ + # g 鍝嶅簲褰㈢姸 (鍗婇ケ鍜岀偣/鍙傝冨) + sharp_x0 = 120.0 # (璁板綍鐢) 鍏ㄥ眬娓呮櫚搴﹀崐楗卞拰鐐; g_sharp 鐜版敼鐢 worst_block + worst_x0 = 6.5 # worst_block 鍗婇ケ鍜岀偣銆傝皟涓(鍘5.0)瑕佹眰鏇存竻鏅般傜湡鏈+OCR: 涓嶅彲璇<4.6/鍙>5.5銆 + # g_sharp = worst/(worst+worst_x0): worst=5鏃秅_sharp=0.5, 鐧界洅娓呮櫚鏍囩 + # worst 姝e父(涓嶈鐧藉簳绋閲)鏁呬笉璇垽need_focus銆傞槇鍊煎緟鐧芥爣绛炬爣鍑嗘暟鎹簿鏍° + change_ref = 0.5 # (鏃, 宸插純鐢: 绋冲畾鎬ф敼鐢ㄥ抚闂村儚绱犲樊) + motion_ref = 15.0 # 甯ч棿妯$硦鍚庡钩鍧囧儚绱犲樊杈炬瑙嗕负"瀹屽叏鍦ㄥ姩"銆傛牎鍑嗕緷鎹: OCR璇佸疄鍙鐨 + # 鎴愬垎闈(瀵嗛泦瀛)妯$硦鍚 motion 杈緙8.4 浠嶅彲璇, 椤昏兘杩; 鏁 ref 涓婅皟鍒15, + # 浣垮彲璇诲抚 stable>=0.44 閫氳繃, 鍚屾椂鍓х儓鏅冨姩(>15)浠嶈鎷掋備笅鐣屽緟琛ラ噰鏅冨姩鏍锋湰绮炬牎銆 + edge_x0 = 0.01 # 杈圭紭瀵嗗害鍗婇ケ鍜 (鍖哄垎鏈夊唴瀹/绌虹櫧) + light_lo = 40.0 # 浜害涓嬫部 + light_hi = 220.0 # 浜害涓婃部(杩囨洕) + light_soft = 25.0 # 浜害杞竟瀹藉害 + # 鏉冮噸 (璇箟娓呮櫚, 鍙仛 ablation) + w_sharp = 1.0 + w_stable = 1.0 + w_content = 0.6 + w_light = 0.4 + sharp_gate_gamma = 0.5 # sharp 涔樻ч棬寮哄害 (0=涓嶅惁鍐, 澶=绯婂抚寮虹儓鍘嬪垎) + # 鍐崇瓥闃堝 (浣滅敤鍦ㄥ綊涓鍖栫患鍚堝垎涓, 鍙涓や釜, 杩滃皯浜庡師鏉ヤ竴鍫嗙‖闃堝) + ACCEPT_Q = 0.55 # 鍙敤鎬ч棬: 璋冧弗(鍘0.40澶澗, 鏀捐繃妯$硦鍥)銆0.55 瑕佹眰鏇存竻鏅版墠鏀捐銆 + # 瀹佹嫆缁濅笉蹇甸敊: 閫佷笅娓哥殑蹇呴』澶熸竻鏅, 妯$硦鐨勫畞鍙鐢ㄦ埛閲嶆媿銆 + # 鏀惧鐨勭悊鐢(鏁版嵁寰楀嚭): 鍏ㄥ眬quality鍒嗕笉寮'鑳絆CR'涓'杞诲井绯', + # 閭f槸浠诲姟灞傜殑绮惧垽(灞閮/鏂囧瓧鍖烘竻鏅板害), 閫氱敤灞備笉瓒婁繋浠e簴銆 + FOCUS_G = 0.35 # (淇濈暀) focus_gain 鍙傝 + # 绋冲畾鎬т紭鍏堜笁鍑哄彛鐨勯棬闄 (娈电骇涓綅鏁颁笂鍒ゅ畾; 寰呯湡鏈 rerun 鏍″噯) + STABLE_MIN = 0.35 # 鍑哄彛A: g_stable 浣庝簬姝 = 鎸佺画鍦ㄥ姩 -> "璇锋嬁绋" (绗竴闂)銆 + # 浠0.45闄嶅埌0.35: 閰嶅悎 motion_ref=15, 璁㎡CR璇佸疄鍙鐨勫瘑闆嗗瓧闈㈤氳繃, 淇璇嫆銆 + SEVERE_FLOW = 8.0 # 鍑哄彛A0: 娈电骇鍏夋祦浣嶇Щ(鍍忕礌)>=姝 = 涓ラ噸鏅冨姩涓鍒鍒囨嫆缁濄傚垵鍊8, + # 寰 shake 鏍锋湰鏍″噯("澶氬ぇ浣嶇Щ绠椾弗閲"); 鍏夋祦浣嶇Щ鐗╃悊鍙В閲娿 + LIGHT_MIN = 0.35 # 鍑哄彛D(鐣): g_light 浣庝簬姝や笖涓烘渶宸 -> 澶殫 + CONTENT_MIN = 0.30 # 鍑哄彛E(鐣): g_content 浣庝簬姝 + SHARP_OK_FOR_E = 0.45 # 鍑哄彛E(鐣): 鍒"娌″鍑"瑕佹眰 sharp 澶熼珮(娓呮櫚鍗存病瀛楁墠绠楀閿) + MIN_QUALIFIED = 1 + + +# ============================================================ +# 鍦烘櫙閰嶇疆 (鏍稿績: 鏈哄埗涓濂, 鍙傛暟鎸夊満鏅爣瀹 鈥斺 "鏉冮噸鐢卞満鏅墿鐞嗙壒寰+瀹夊叏绛夌骇鍐冲畾") +# 鑽洅(medicine): 灏忓瓧/楂樺嵄/鐧藉簳 -> 鍋忔嫆缁濄佽姹傛竻鏅般侀浂瀹瑰繊銆= 榛樿 FunnelConfig銆 +# 鏂囧叿(stationery): 涓ぇ瀛/浣庡嵄/褰╄壊鍖呰 -> 鍙斁鏉俱佸厑璁告洿澶氭斁琛屻 +# 鍥烘湁鏈哄埗(瀵圭劍/澶氬抚绛涢/鍊掔疆/鍏夋祦鏅冨姩)瀹屽叏涓鑷, 鍙湁涓嬮潰杩欎簺鍒ゅ畾鍙傛暟涓嶅悓銆 +# ============================================================ +def make_scene_config(scene: str = "medicine") -> "FunnelConfig": + """鎸夊満鏅繑鍥炴爣瀹氬ソ鐨 FunnelConfig銆傛満鍒朵唬鐮佷笉鍙, 鍙敼鍙傛暟銆""" + cfg = FunnelConfig() + if scene in ("medicine", "鑽洅", "drug"): + return cfg # 鑽洅 = 鐜版湁榛樿(灏忓瓧/楂樺嵄/涓) + if scene in ("stationery", "鏂囧叿", "supplies", "鐢熸椿鐢ㄥ搧"): + # 鐢熸椿鐢ㄥ搧/鏂囧叿: 瀹夊叏绛夌骇浣(蹇甸敊鍏崇郴涓嶅ぇ) -> 鏀捐闃堝兼斁鏉, 灏戞嫆缁濄佸皯鎵撴壈銆 + # 鍙傛暟渚濇嵁: 缃戠粶鎽勫儚澶 clip_windows csv (routing 鍒嗘瀽) + 瀹夊叏绛夌骇銆 + cfg.worst_x0 = 5.0 # 鏉(鑽洅6.5): 鎷︽埅鍒ゆ嵁, 鑽洅涓ユ枃鍏锋澗(瀹夊叏绛夌骇浣)銆 + # 鎵嬪姩瀹: csv閲 worst_block 瀵筼k/blur鍖哄垎寰堝急(鍥鹃兘杈冩竻鏅, + # AUC~0.52), 绠椾笉鍑哄彲闈犲, 鎸"鏂囧叿姣旇嵂鐩掓澗"鎵嬪姩瀹氭澗涓妗c + cfg.ACCEPT_Q = 0.48 # 鐣ユ斁鏉(鑽洅0.55): 瀹夊叏浣, 瀹归敊楂樹竴鐐广 + cfg.SEVERE_FLOW = 8.0 # 娌跨敤鑽洅: csv 閲 seg_flow 鏄渶寮哄垽鎹(AUC~0.71), >8 鍚 + # ok鐜囬闄, 涓庤嵂鐩掍竴鑷淬傛檭鍔ㄦ槸閫氱敤鐗╃悊绾︽潫, 涓嶉殢鍦烘櫙鏀炬澗銆 + return cfg + # 鏈煡鍦烘櫙 -> 閫鍥炶嵂鐩掗粯璁, 骞舵彁绀 + print(f"[scene] 鏈煡鍦烘櫙 '{scene}', 鐢ㄨ嵂鐩掗粯璁ら厤缃") + return cfg + + +class FunnelResult: + def __init__(self): + self.accepted = False + self.need_focus = False # 绋充絾绯: 涓婂眰搴旇Е鍙慉F鍐嶉噰涓鎵 + self.reject_reason = "" + self.best_index = -1 + self.best_jpg = None + self.per_frame = [] + self.motions = [] # 杩欓噷瀛 rel_change 搴忓垪 + self.components = {} # 娈电骇鍚 gi 鍒嗛噺 (褰掑洜鐢) + self.timings = {} + + def to_dict(self): + def _fmt(v): + if isinstance(v, (int, float)): + return round(v, 1) + return v # list 涔嬬被鍘熸牱 (濡 grab_per_frame_ms) + return { + "accepted": self.accepted, + "need_focus": self.need_focus, + "reject_reason": self.reject_reason, + "best_index": self.best_index, + "per_frame": self.per_frame, + "rel_changes": [round(m, 3) for m in self.motions], + "components": self.components, + "timings": {k: _fmt(v) for k, v in self.timings.items()}, + } + + +def run_funnel(frames: list[bytes], cfg: FunnelConfig = FunnelConfig()) -> FunnelResult: + """璇勫垎鍑芥暟鐗堟紡鏂: + 1) 姣忓抚绠 metrics(sharp/浜害/edge) + 2) 甯ч棿鐩稿鍙樺寲 -> 姣忓抚 quality + focus_gain (杩炵画鍒, 闈炵‖闃堝) + 3) 鍐崇瓥: max(quality)>=ACCEPT_Q -> 閫夎甯(閫変紭); + 鍚﹀垯 max(focus_gain)>=FOCUS_G -> 寤鸿瀵圭劍(need_focus); + 閮戒綆 -> 鎷掔粷銆 + 杩斿洖閲屽甫 need_focus 鏍囧織, 渚涗笂灞傚喅瀹"绋充絾绯->瑙﹀彂AF->鍐嶉噰"銆""" + r = FunnelResult() + if not frames: + r.reject_reason = "no_frames" + return r + + # 1) 鍗曞抚 metrics + t0 = time.monotonic() + metrics_list = [frame_metrics(j) for j in frames] + r.timings["level1_metrics_ms"] = (time.monotonic() - t0) * 1000 + + # 2) 甯ч棿"杩愬姩"(绋冲畾鎬ц緭鍏) + 姣忓抚璇勫垎 + # 绋冲畾鎬х敤鐪熸鐨勫抚闂村儚绱犲樊鍒(frame_motion)琛¢噺 鈥斺 浣嶇Щ/杞姩浼氭敼鍙樼敾闈㈠唴瀹, + # 浣嗕笉涓瀹氭敼鍙樻竻鏅板害, 鏁呬笉鑳界敤 sharp 鍙樺寲浠f浛銆傚綊涓鍖栧埌 [0,1] 鐨 rel銆 + t1 = time.monotonic() + per = [] + prev_jpg = None + flow_mags = [] # 鐩搁偦甯у厜娴佷綅绉诲箙搴(涓ラ噸鏅冨姩涓鍒鍒囩敤) + for i, m in enumerate(metrics_list): + if prev_jpg is None: + motion = 0.0 + else: + motion = frame_motion(prev_jpg, frames[i]) # 骞冲潎缁濆鍍忕礌宸 0-255 + flow_mags.append(optical_flow_motion(prev_jpg, frames[i])) # 鍏夋祦浣嶇Щ(鍍忕礌) + prev_jpg = frames[i] + rel = min(1.0, motion / cfg.motion_ref) # 褰掍竴: motion>=motion_ref 瑙嗕负瀹屽叏鍦ㄥ姩 + sc = frame_score(m, rel, cfg) + per.append({**m, "rel_change": round(rel, 3), "motion": round(motion, 2), **sc}) + r.motions.append(rel) + r.per_frame = per + # 娈电骇鍏夋祦: 鐢ㄤ腑浣嶆暟鎶楀崟甯у櫔澹般傚ぇ = 鏁存鐩告満鍦ㄥぇ骞呯Щ鍔 = 涓ラ噸鏅冨姩銆 + seg_flow = float(sorted(flow_mags)[len(flow_mags) // 2]) if flow_mags else 0.0 + r.timings["level2_score_ms"] = (time.monotonic() - t1) * 1000 + + # 3) 鍐崇瓥: 绋冲畾鎬т紭鍏堢殑涓夊嚭鍙 (A/B/C), argmin(gi) 褰掑洜, D/E 鐣欐帴鍙 + t2 = time.monotonic() + qualities = [p["quality"] for p in per] + # 娈电骇鑱氬悎鍚勫垎閲 (鐢ㄤ腑浣嶆暟鎶楀崟甯у櫔澹) + def med(key): + vals = sorted(p[key] for p in per) + n = len(vals) + return vals[n // 2] if n % 2 else (vals[n // 2 - 1] + vals[n // 2]) / 2 + seg_stable = med("g_stable") + seg_sharp = med("g_sharp") + seg_content = med("g_content") + seg_light = med("g_light") + # best 甯ч夋嫨: 鍙寜銆愭竻鏅板害銆戦夋渶鑳借鐨勪竴甯, 涓嶇敤缁煎悎 quality銆 + # 鍘熷洜: quality 鍚 stable 鍒嗛噺, 浼氳"鏅冨姩鏌愮灛闂存伆濂藉抚闂磎otion浣(杞姌鐐)浣嗙敾闈㈢硦"鐨勫抚 + # quality 铏氶珮, 鍘嬭繃"绋嶅姩浣嗘竻鏅板彲璇"鐨勫抚 -> best 閫夋垚绯婄殑(鐪熸満楠岃瘉: f0绯婅閫/f2娓呮櫚娌¢)銆 + # 蹇靛瓧鑳戒笉鑳借鍙彇鍐充簬娓呮櫚搴, 涓庤甯х灛鏃剁ǔ涓嶇ǔ鏃犲叧; 绋冲畾鎬у彧鐢ㄤ簬鏁存鏄惁鎷掔粷(鍑哄彛A)銆 + # 鐢 worst_block(鏈绯婃湁鍐呭鍧)閫: 閫"鏈绯婂涔熸渶娓呮櫚"鐨勯偅甯 = 鏈鍙銆 + def _clarity(i): + return per[i].get("worst_block", per[i].get("sharpness", 0.0)) + best_q = max(range(len(per)), key=_clarity) + # quality 浠嶄繚鐣欎緵鍑哄彛B鐨勬帴鍙楅槇鍊煎垽鏂(涓嬮潰 qualities[best_q]>=ACCEPT_Q) + r.timings["level3_decide_ms"] = (time.monotonic() - t2) * 1000 + + # 褰掑洜鍒嗛噺琛 (渚 argmin 璺敱 + 璁板綍) + r.components = {"stable": round(seg_stable, 3), "sharp": round(seg_sharp, 3), + "content": round(seg_content, 3), "light": round(seg_light, 3), + "flow": round(seg_flow, 2)} # 娈电骇鍏夋祦浣嶇Щ(鍍忕礌) + + # ---- 鍑哄彛 A0: 涓ラ噸鏅冨姩涓鍒鍒 鈥斺 鍏夋祦浣嶇Щ杩囧ぇ = 鐩告満澶у箙绉诲姩, 涓嶇湅鍒殑鐩存帴鎷 ---- + # 鍙В閲: 鐢婚潰鏁存骞冲潎绉诲姩 >SEVERE_FLOW 鍍忕礌 = 鏄庢樉鍦ㄦ檭, 蹇靛瓧蹇呯硦, 璁╃敤鎴峰仠绋炽 + # 鍏夋祦姣旈愬儚绱犲樊椴佹(涓嶈瀵嗛泦瀛椾吉杩愬姩楠); 闃堝煎緟 shake 鏍锋湰鏍″噯銆 + if seg_flow >= cfg.SEVERE_FLOW: + r.reject_reason = "severe_shake" # 鍑哄彛 A0: "鏅冨緱鍘夊, 璇锋嬁绋" + r.need_focus = False + r.best_index = best_q + return r + + # ---- 鍑哄彛 A: 绋冲畾鎬т紭鍏 鈥斺 鎸佺画鍦ㄥ姩, 涓嶇湅娓呮櫚搴, 鐩存帴璇锋嬁绋 ---- + # (鍔ㄧ殑鏃跺欏鐒︿篃娌$敤, 涓旂敤鎴锋病鑰佸疄, 涓嶈娴垂鏃堕棿鍘绘寫鍥) + if seg_stable < cfg.STABLE_MIN: + r.reject_reason = "unstable" # 鍑哄彛 A: "璇锋嬁绋/瀵瑰噯" + r.need_focus = False + r.best_index = best_q # 浠呬緵鍙傝, 涓嶅杺妯″瀷 + return r + + # ---- 绋冲畾銆傜湅鏈夋病鏈夊抚澶熸竻鏅 ---- + if qualities[best_q] >= cfg.ACCEPT_Q: + # ---- 鍑哄彛 B: 绋 + 鏈夋竻鏅板抚 -> 缁欐渶娓呮櫚鐨 ---- + r.accepted = True + r.best_index = best_q + r.best_jpg = frames[best_q] + r.need_focus = False + return r + + # ---- 绋充絾娌℃湁澶熸竻鏅扮殑甯: 褰掑洜 argmin, 鍐冲畾鏄 C(瀵圭劍) 杩樻槸 D/E ---- + # 鍦"绋冲畾"鍓嶆彁涓, 鐪嬪摢涓垎閲忔渶鎷栫疮: sharp浣=澶辩劍(C); light浣=鏆(D); + # content浣庝絾sharp楂=娓呮櫚鍗存病鍐呭=娌″鍑(E)銆俛rgmin 杩炵画璺敱, 闈炵‖鍫嗛槇鍊笺 + cand = {"sharp": seg_sharp, "light": seg_light, "content": seg_content} + worst = min(cand, key=cand.get) + + if worst == "light" and seg_light < cfg.LIGHT_MIN: + # 鍑哄彛 D (鐣欐帴鍙, 鏈縺娲诲共棰): 澶殫 -> 鏈潵 CLAHE 鏁板瓧澧炲己 / 鎻愮ず鍒颁寒澶 + r.reject_reason = "too_dark" # TODO: 鎺 CLAHE 鍚庢敼涓哄彲鏁 + r.need_focus = False + r.best_index = best_q + return r + + # 鍑哄彛 E锛坅imed_wrong锛夊凡寮冪敤锛氬師璁捐锛-v/OCR锛夌敤"娓呮櫚浣嗘棤鍐呭缁撴瀯"鍒"娌″鍑"锛 + # 鐩殑鏄帓闄よ儗鏅共鎵般備絾 -o 涓嶅彈鑳屾櫙褰卞搷銆佸畬鍏ㄦ病瀛楁椂浼氳佸疄璇"娌$湅鍒板瓧"锛 + # 涓嶉渶瑕佹紡鏂楁嫤銆傛晠杩欑甯х洿鎺ユ斁琛岋紝浜ょ粰妯″瀷鑷繁鍒ゆ柇锛岄伩鍏嶈鍒ゆ墦鏂 + # 锛堢湡姝e嵄闄╃殑"鍍忓瓧鍙堜笉鏄瓧"鏄彟涓涓棶棰橈紝寰 720p 鏁版嵁鐢ㄥ埆鐨勪俊鍙烽噸鍋氥傦級 + if worst == "content" and seg_sharp >= cfg.SHARP_OK_FOR_E and seg_content < cfg.CONTENT_MIN: + r.accepted = True + r.best_index = best_q + r.best_jpg = frames[best_q] + r.need_focus = False + return r + + # ---- 鍑哄彛 C: 绋充絾绯 (sharp 涓诲鎷栫疮) -> 瑙﹀彂AF閲嶉噰 ---- + r.accepted = False + r.need_focus = True + r.reject_reason = "need_focus" # 涓婂眰瑙﹀彂AF+閲嶉噰, 浠嶄笉琛屾墠鏈缁堟嫆缁 + r.best_index = best_q + return r + + +# ============================================================ +# 鏂瑰悜澶勭悊 & 鎺ㄧ悊 鈥斺 鐣欐々, 涓嬩竴姝ュ~ +# ============================================================ +def _center_crop(gray, frac=0.6): + """鍙栫敾闈腑蹇冨尯, 鎺掓帀鍥涘懆鑳屾櫙(鎵/妗岄潰/澧)銆傜敤鎴蜂細鎶婄洰鏍囧ぇ鑷村鍑嗕腑蹇, + 鏁呬腑蹇冨尯鏇村彲鑳芥槸鏂囧瓧涓讳綋; 鍦ㄥ叾涓婄畻鍒ゆ嵁鍙樉钁楀噺杞昏儗鏅薄鏌撱""" + h, w = gray.shape[:2] + y0, y1 = int(h * (1 - frac) / 2), int(h * (1 + frac) / 2) + x0, x1 = int(w * (1 - frac) / 2), int(w * (1 + frac) / 2) + return gray[y0:y1, x0:x1] + + +def _binv_fg(gray): + _, b = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU) + return (b > 0).astype(np.float64) + + +def _detect_sideways(gray): + """90掳/270掳渚у悜: 姘村钩鎶曞奖鏂瑰樊 vs 鍨傜洿鎶曞奖鏂瑰樊銆傚妗d腑蹇冨尯(0.5/0.6/0.7)鎶曠エ, + 姣斿崟涓姣斾緥鏇寸ǔ(鏂囧瓧涓嶅湪姝d腑蹇冩椂涓嶆槗琚崟妗h儗鏅薄鏌撳甫鍋)銆 + 杩斿洖 (verdict, ratio): 'upright'/'sideways'/'ambiguous'銆俽atio 鍙栦腑浣嶆暟銆""" + ratios = [] + for frac in (0.5, 0.6, 0.7): + b = _binv_fg(_center_crop(gray, frac)) + hvar = float(np.var(b.sum(axis=1))) + vvar = float(np.var(b.sum(axis=0))) + ratios.append(hvar / (vvar + 1e-6)) + ratios.sort() + med = ratios[1] # 涓綅鏁, 鎶楀崟妗e紓甯 + votes = ["sideways" if r < 0.67 else ("upright" if r > 1.5 else "ambiguous") + for r in ratios] + # 澶氭暟绁: 3妗i噷鈮2妗d竴鑷存墠瀹, 鍚﹀垯 ambiguous + for v in ("sideways", "upright"): + if votes.count(v) >= 2: + return v, round(med, 2) + return "ambiguous", round(med, 2) + + +def _detect_upside_down(gray): + """180掳姝e: '娉ㄦ按瀹归噺'鍒ゆ嵁(璧勬枡鏂规硶, 鐪熷浘楠岃瘉瀵逛腑鏂囨湁鏁)銆 + 瀵规瘡鍒: 鏈楂樻枃瀛楃偣涓婃柟绌虹櫧(top_cap) vs 鏈浣庢枃瀛楃偣涓嬫柟绌虹櫧(bot_cap)銆 + 姝g珛鐗堥潰鏂囧瓧鏁翠綋涓嬫柟鐣欑櫧澶 -> (bot-top)>0; 鍊掔疆缈昏浆 -> <0銆 + **鐪嬬増闈㈢骇涓婁笅鐣欑櫧, 涓嶄緷璧栧崟瀛楀绉版, 缁曞紑涓枃瀛楀绉版缁撱** + 鐪熷浘楠岃瘉: 姝g珛+0.66/鍊掔疆-0.48, 鎶椔8掳鍊炬枩銆佹姉瑁佸壀姣斾緥(0.5~0.9)銆傚繀椤昏鍓(鏁村浘浼氳鑳屾櫙娣)銆 + 杩斿洖 (verdict, score): 'upright'/'flipped'/'ambiguous'銆""" + b = _binv_fg(_center_crop(gray, 0.7)) + h, w = b.shape + tc = bc = cnt = 0 + for x in range(0, w, 3): + col = np.where(b[:, x] > 0)[0] + if len(col) < 2: + continue + tc += col[0] + bc += (h - 1 - col[-1]) + cnt += 1 + if cnt < 10: + return "ambiguous", 0.0 + score = (bc - tc) / (bc + tc + 1e-9) + if score > 0.15: + return "upright", round(score, 3) + if score < -0.15: + return "flipped", round(score, 3) + return "ambiguous", round(score, 3) + + +def _detect_truncation(gray, margin=10): + # 鍏堢‘璁や腑蹇冨尯鏈夋枃瀛椾富浣; 鍚﹀垯涓嶅垽鎴柇(閬垮厤鑳屾櫙杈圭紭璇姤) + cb = _binv_fg(_center_crop(gray, 0.6)) + if cb.mean() < 0.01: + return {}, {} # 涓績娌″唴瀹, 涓嶈皥鎴柇 + b = _binv_fg(gray) + edges = {"top": b[:margin, :].mean(), "bottom": b[-margin:, :].mean(), + "left": b[:, :margin].mean(), "right": b[:, -margin:].mean()} + overall = b.mean() + if overall < 1e-4: + return {}, {} + # 璐磋竟鍒ゆ嵁鏀剁揣: 杈圭紭瀵嗗害闇 >= 鏁村浘鐨勪竴瀹氭瘮渚, 涓 >= 涓績瀵嗗害鐨勪竴瀹氭瘮渚 + # (鎺掓帀"鑳屾櫙绾圭悊璐磋竟浣嗕腑蹇冩墠鏄枃瀛"鐨勮鎶) + center_d = cb.mean() + touched = {k: (v >= overall * 0.6 and v >= center_d * 0.4 and v > 0.02) + for k, v in edges.items()} + return {k: round(v, 4) for k, v in edges.items()}, {k: t for k, t in touched.items() if t} + + +_PAN_HINT = {"top": "璇峰悜涓婄湅涓鐐", "bottom": "璇峰悜涓嬬湅涓鐐", + "left": "璇峰悜宸︾湅涓鐐", "right": "璇峰悜鍙崇湅涓鐐"} + + +_ORI_CLS = None # 鏂瑰悜鍒嗙被鍣ㄥ崟渚(鍏ㄥ眬鍙姞杞戒竴娆) +_ORI_CLS_TRIED = False # 鏄惁宸插皾璇曞姞杞(閬垮厤鍙嶅閲嶈瘯澶辫触) +_ORI_CLS_ERR = "" # 鍔犺浇澶辫触鍘熷洜(渚涗笂灞傚瀹炴姤鍛婏紝涓嶈鍙湅"棰勭儹灏辩华") + + +def orient_classifier_status(): + """杩斿洖 (鏄惁鍙敤, 澶辫触鍘熷洜)銆 + + 涓轰粈涔堥渶瑕佸畠锛歚[棰勭儹] 鏂瑰悜鍒嗙被鍣ㄥ氨缁猔 閭h鏄**鏃犳潯浠舵墦鍗拌楁椂**鐨勶紝 + 鍔犺浇澶辫触鏃 process_orientation 浼氶潤榛樺洖閫鍒拌交閲 CV 鍒ゆ嵁銆佺収鏍疯繑鍥烇紝 + 棰勭儹鐓ф牱鎶"灏辩华" 鈥斺 浜庢槸"棰勭儹鎴愬姛"鏍规湰涓嶈兘璇佹槑鍒嗙被鍣ㄥ彲鐢ㄣ + 瀹炴祴 panel 鐜涓嬪姞杞藉け璐ワ紙paddle 寰幆瀵煎叆锛夛紝閫鍒拌交閲忓垽鎹悗 + 绗竴杞氨璇垽 orient_flipped锛岃屾墜鏁茬幆澧冨姞杞芥垚鍔熴佸垽瀹氭甯搞 + """ + return (_ORI_CLS is not None), _ORI_CLS_ERR + +def _get_orient_classifier(): + """鎳掑姞杞 PaddleOCR 鏂囨。鏂瑰悜鍒嗙被鍣(PP-LCNet, 6.75MB, 0/90/180/270)銆 + 鍗曚緥: 鍙湪棣栨璋冪敤鏃跺姞杞戒竴娆(鍑犵), 涔嬪悗姣忔鎺ㄧ悊浠呭嚑ms銆 + 闄嶇骇: 鏈畨瑁/鍔犺浇澶辫触 -> 杩斿洖 None, process_orientation 鍥為鍒拌交閲廋V鍒ゆ嵁, 涓嶅穿銆""" + global _ORI_CLS, _ORI_CLS_TRIED + if _ORI_CLS is not None: + return _ORI_CLS + if _ORI_CLS_TRIED: + return None # 涔嬪墠璇曡繃涓斿け璐, 涓嶅啀閲嶈瘯 + _ORI_CLS_TRIED = True + # 鍔犺浇鏂瑰悜鍒嗙被鍣ㄣ備紭鍏堢敤鏈绠鍗曠殑鍐欐硶(瀹炴祴 SmartGlasses 鐜鍙敤, 涓 torch 涓嶅啿绐 鈥斺 + # 褰撳垵鍐茬獊鐨勬槸瀹屾暣 PaddleOCR-GPU 璇嗗埆, 涓嶆槸杩欎釜杞婚噺鏂瑰悜鍒嗙被鍣)銆 + # 鑻ユ兂鏄惧紡鎸囧畾璁惧, 鍚庨潰鐨勫弬鏁板彉浣撲綔涓哄閫変緷娆″皾璇曘 + from paddleocr import DocImgOrientationClassification + last_err = None + for kw in (dict(model_name="PP-LCNet_x1_0_doc_ori"), # 鏈绠鍗(鍘熸潵鑳界敤鐨) + dict(model_name="PP-LCNet_x1_0_doc_ori", device="cpu"), # 鏄惧紡CPU(鍙) + dict()): # 鍏ㄩ粯璁ゅ厹搴 + try: + _ORI_CLS = DocImgOrientationClassification(**kw) + print(f"[orient] PaddleOCR 鏂瑰悜鍒嗙被鍣ㄥ凡鍔犺浇 (PP-LCNet, 鍙傛暟={list(kw) or 'default'})") + return _ORI_CLS + except Exception as e: + last_err = e + continue # 杩欎釜鍙傛暟缁勫悎涓嶈, 鎹笅涓涓 + global _ORI_CLS_ERR + _ORI_CLS_ERR = f"{type(last_err).__name__}: {last_err}" + print(f"[orient] 鏂瑰悜鍒嗙被鍣ㄥ姞杞藉け璐, 鍥為鍒拌交閲廋V鍒ゆ嵁: {last_err}") + _ORI_CLS = None + return None + + +# 鏂囨。鏂瑰悜鏍囩(鍥惧儚琚『鏃堕拡鏃嬭浆鐨勮搴) -> (鐘舵, 缁欑敤鎴风殑杞悜鎻愮ず) +# 娉ㄦ剰: "鍥惧儚鏃嬭浆瑙"涓"鐢ㄦ埛杞嵂鐩掓柟鍚"鐩稿弽銆傛枃妗堝厛鎸夋爣鍑嗗畾涔, 瀹炴満鏍稿鍚庡浐瀹氥 +# 90掳: 鐢婚潰椤烘椂閽堣浆浜90掳 = 鑽洅琚嗘椂閽堟斁浜 -> 鐢ㄦ埛搴旈『鏃堕拡杞洖? 闇瀹炴満鏍稿, 鏁呬繚鐣 raw 鏍囩璋冭瘯銆 +_ORIENT_MAP = { + "0": ("upright", None), + "90": ("sideways", "璇锋妸鑽洅椤烘椂閽堣浆90掳"), + "270": ("sideways", "璇锋妸鑽洅閫嗘椂閽堣浆90掳"), + "180": ("flipped", "鑽洅鎷垮弽浜嗭紝璇蜂笂涓嬬炕杞"), +} + + +def _classify_orientation(img_bgr): + """鐢ㄥ垎绫诲櫒鍒ゆ柟鍚戙傝繑鍥 (raw_label, conf) 鎴 (None, 0) 鑻ヤ笉鍙敤/浣庣疆淇°""" + cls = _get_orient_classifier() + if cls is None: + return None, 0.0 + try: + res = cls.predict(img_bgr) # 浼 ndarray, 鍏嶈惤鐩 + r = res[0] + labels = r.get("label_names") or [str(x) for x in r.get("class_ids", [])] + scores = r.get("scores", [0.0]) + raw = str(labels[0]).replace("_degree", "").strip() + conf = float(max(scores)) if scores else 0.0 + return raw, conf + except Exception as e: + print(f"[orient] 鎺ㄧ悊澶辫触: {e}") + return None, 0.0 + + +def process_orientation(best_jpg: bytes): + """绗2绾(鏂瑰悜) + 绗3绾(鍙栨櫙瀹屾暣) 鈥斺 鍙娓呮櫚鐨 best 甯у仛涓娆° + 鏂瑰悜: 浼樺厛鐢 PaddleOCR 杞婚噺鏂瑰悜鍒嗙被鍣(PP-LCNet, 鐪熷浘楠岃瘉0/90/180/270鍏ㄥ噯, 缃俊~0.92); + 鏈/浣庣疆淇 -> 鍥為杞婚噺CV鍒ゆ嵁銆180掳璧板甯т竴鑷存('璇风‘璁ゆ鍙'鍦╤andler瑙﹀彂)銆 + 杩斿洖 (jpg, diag)銆俤iag 鍚柟鍚戝弽棣 + 鎴柇鍙嶉 + incomplete 鏍囧織(浼犱笅娓竝rompt)銆 + [鐣欐々] 鍦嗘煴鏇查潰鐭 = 鐙珛璇鹃, 涓嶅湪姝ゅ銆""" + arr = np.frombuffer(best_jpg, np.uint8) + img = cv2.imdecode(arr, cv2.IMREAD_COLOR) + if img is None: + return best_jpg, {"ok": False} + gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) + + orient_hint = None + orient_state = "upright" + ud_score = 0.0 + side_ratio = "" + orient_raw = "" + orient_conf = 0.0 + CONF_MIN = 0.60 # 缃俊闂ㄦ帶: 浣庝簬姝や笉纭垽, 璧皍ncertain/鍏滃簳 + + raw, conf = _classify_orientation(img) + orient_raw, orient_conf = (raw or ""), conf + if raw is not None and conf >= CONF_MIN and raw in _ORIENT_MAP: + # 鍒嗙被鍣ㄥ彲鐢ㄤ笖缃俊瓒冲 -> 鐩存帴鎸夋槧灏勭粰鐘舵+鎻愮ず + # (瀹炴祴180 conf~0.93寰堢‘瀹, 涓嶄細涓0娣锋穯, 鏁呭嵆鏃舵彁绀; 鑻ラ渶婊ゅ櫔澹板湪绨囧唴澶氬抚鎶曠エ灞傚仛) + orient_state, orient_hint = _ORIENT_MAP[raw] + elif raw is not None: + # 鍒嗙被鍣ㄧ粰浜嗙粨鏋滀絾浣庣疆淇 -> 涓嶇‖鍒 + orient_state = "uncertain" + else: + # 鍒嗙被鍣ㄤ笉鍙敤 -> 鍥為杞婚噺CV鍒ゆ嵁(鏁村潡鎶曞奖/娉ㄦ按瀹归噺) + side, side_ratio = _detect_sideways(gray) + if side == "sideways": + orient_state, orient_hint = "sideways", "璇锋妸鑽洅杞90掳" + else: + ud, ud_score = _detect_upside_down(gray) + orient_state = {"flipped": "flipped", "upright": "upright"}.get(ud, "uncertain") + + # 鍙栨櫙瀹屾暣鎬(鎴柇) 鈥斺 鏄剧ず灞傛殏鍏, 鎺ュ彛淇濈暀 + edges, truncated = _detect_truncation(gray) + pan_hint = None + if truncated: + worst = max(truncated, key=lambda k: edges.get(k, 0)) + pan_hint = _PAN_HINT.get(worst) + + diag = { + "ok": True, + "orient_state": orient_state, # upright/sideways/flipped/uncertain + "orient_hint": orient_hint, # 渚у悜鍗虫椂鎻愮ず; 180鐨"璇风‘璁"鐢卞甯т竴鑷存у湪handler鍔 + "orient_raw": orient_raw, # 鍒嗙被鍣ㄥ師濮嬫爣绛(0/90/180/270), 瀹炴満鏍稿杞悜鐢 + "orient_conf": round(orient_conf, 3), + "sideways_ratio": side_ratio, + "flipped_frame": orient_state == "flipped", # 渚涘甯т竴鑷存х疮绉 + "ud_score": ud_score, + "truncated_edges": list(truncated.keys()), + "pan_hint": pan_hint, + "incomplete": bool(truncated), # 浼犱笅娓竝rompt: 鍙康鍙閮ㄥ垎 + } + return best_jpg, diag # 绗竴鐗堜笉鑷姩杞, 鍙瘖鏂+鎻愮ず(鏂瑰悜鐢辩敤鎴疯皟) + + +def infer(frame_jpg: bytes) -> Optional[str]: + """[鐣欐々] 鏈潵: 鎶 best 甯у杺缁 -v (鎴 Qwen-VL) 鍋氬康瀛/璇嗗埆銆 + 鐜板湪杩斿洖 None銆傛帴鎺ㄧ悊鏃舵妸杩欓噷鎹㈡垚鐪熺殑妯″瀷璋冪敤銆""" + return None + + +# ============================================================ +# CV 鎸囨爣: 鎷夋櫘鎷夋柉鏂瑰樊(娓呮櫚搴) + 骞冲潎浜害 +# 杩欎袱涓氨鏄笁绾ф紡鏂椾竴绾х殑鍒ゆ嵁; 鍦ㄨ繖閲屽疄鏃舵樉绀轰互鏍囧畾闃堝笺 +# ============================================================ +def frame_metrics(jpg: bytes) -> dict: + """杩斿洖涓甯х殑 CV 鎸囨爣瀛楀吀銆傝褰曞櫒浼氬姩鎬佹妸鎵鏈 key 褰撲綔 CSV 鍒 鈥斺 + 浠ュ悗 CV 閭f鍔犳柊鎸囨爣(濡傚抚闂村樊鍒嗘祴鏅冨姩銆佹柟鍚戞娴嬪垎), 鍙渶鍦ㄨ繖閲屽線 + dict 閲屽姞 key, CSV 鑷姩澶氬垪, 璁板綍浠g爜涓嶇敤鏀广""" + arr = np.frombuffer(jpg, dtype=np.uint8) + img = cv2.imdecode(arr, cv2.IMREAD_GRAYSCALE) + if img is None: + return {"sharpness": 0.0, "brightness": 0.0, "kb": round(len(jpg) / 1024, 1)} + lap_var = cv2.Laplacian(img, cv2.CV_64F).var() + edges = cv2.Canny(img, 80, 160) + edge_density = float((edges > 0).mean()) + + # --- 闄勫姞鎸囨爣: 鍏堣褰曚笉鍒ゅ畾, 閲囨暟鎹悗鐢 ok/blur 鏍囨敞鐪嬪摢涓渶鑳藉榻愬彲璇绘 --- + # local_sharp: N脳N 鍒嗗潡鍙栨渶娓呮櫚鍧椼傛姄"灞閮ㄦ湁娓呮櫚鍖"; 浣嗕細琚珮棰戣儗鏅(鏈ㄧ汗)楠楅珮銆 + ls = _local_sharp(img, grid=4) + # text_sharp / text_ratio: MSER 绫绘枃瀛楀欓夊尯鐨勬竻鏅板害 + 闈㈢Н鍗犳瘮銆 + # text_ratio 璇"鏈夋病鏈夊儚瀛楃殑鍖哄煙"(楂橀鑳屾櫙 ratio鈮0, 楠椾笉杩囧畠); + # 浣嗙硦鍒颁竴瀹氱▼搴 MSER 鎵句笉鍒板瓧 -> ratio鈫0 (鑷垜鐭涚浘, 宸茬绾胯瘉瀹)銆 + # 涓 local_sharp 缁勫悎: local楂+ratio浣=娓呮櫚浣嗛潪瀛; 涓よ呯殕浣=绯娿 + ts, tr = _text_sharp(img) + # worst_block / block_var: 鍖归厤"涓澶勭硦鍗充笉鍙敤(OCR 鏍囧噯)"鐨勬偛瑙傚垽鎹 + # worst_block = 鏈夊唴瀹圭殑鍧楅噷鏈绯婄殑娓呮櫚搴 (鐪嬫渶宸, 闈炴渶濂); + # block_std = 鍧楅棿娓呮櫚搴︽爣鍑嗗樊 (澶 = 閮ㄥ垎娓呮櫚閮ㄥ垎绯 = 灞閮ㄧ硦)銆 + # 瀵 OCR: 鍐冲畾鎴愯触鐨勬槸鏈绯婄殑鏂囧瓧鍖, 涓嶆槸鏈娓呮櫚鐨 鈥斺 鏁呯湅 worst 鑰岄潪 best銆 + wb, bstd = _block_sharp_stats(img, grid=4) + # contrast / dark_ratio: 鍖哄垎"绯"涓"鏆/浣庡姣"銆傛殫+浣庡姣旀椂鍏ㄥ眬sharp鍋囨у亸浣, + # 浣 contrast 浣庝細鎻ず鐪熷洜 -> 瀵瑰簲"鏁板瓧澧炲己(CLAHE)"鑰岄潪"瀵圭劍"杩欐潯骞查璺緞銆 + contrast = float(img.std()) + dark_ratio = float((img < 50).mean()) + + # [CLAHE 鎺ュ彛棰勭暀] 鏈潵鏆楃幆澧冨康瀛: 鑻 contrast 浣 / dark_ratio 楂, 鍙湪姝ゅ img + # 鍋 CLAHE/gamma 澧炲己鍚庨噸绠楁寚鏍, 浣滀负"鏁板瓧灞傚共棰"(鍖哄埆浜庡鐒︾殑"鐗╃悊灞傚共棰")銆 + # 鐜板湪涓嶅疄鐜 鈥斺 寰呮湁鏆楃幆澧冩祴璇曟潯浠跺啀濉傜ず鎰: + # if contrast < TH: img_enh = cv2.createCLAHE(...).apply(img); 閲嶇畻 lap_var ... + + return { + "sharpness": round(float(lap_var), 1), # 鍏ㄥ眬鎷夋櫘鎷夋柉鏂瑰樊 + "brightness": round(float(img.mean()), 1), + "edge_density": round(edge_density, 4), + "local_sharp": round(float(ls), 1), # 鏈娓呮櫚鍧 (涔愯, 璁板綍) + "worst_block": round(float(wb), 1), # 鏈绯婄殑鏈夊唴瀹瑰潡 (鎮茶, 鍖归厤OCR鏍囧噯) + "block_std": round(float(bstd), 1), # 鍧楅棿娓呮櫚搴td (澶=灞閮ㄧ硦) + "text_sharp": round(float(ts), 1), # 鏂囧瓧鍖烘竻鏅板害 (璁板綍) + "text_ratio": round(float(tr), 4), # 鏂囧瓧鍖洪潰绉崰姣 (璁板綍) + "contrast": round(contrast, 1), # 瀵规瘮搴(std) (璁板綍) + "dark_ratio": round(dark_ratio, 4), # 鏆楀儚绱犲崰姣 (璁板綍) + "kb": round(len(jpg) / 1024, 1), + } + + +def _block_sharp_stats(gray, grid: int = 4): + """鍒嗗潡娓呮櫚搴︾殑 (鏈绯婃湁鍐呭鍧, 鍧楅棿std)銆 + 鍙粺璁"鏈夊唴瀹"鐨勫潡(edge 闈炴瀬浣), 閬垮厤绾儗鏅潡骞叉壈; + worst = 鏈夊唴瀹瑰潡閲屾渶浣庢竻鏅板害 (瀵瑰簲'涓澶勭硦鍗冲け璐'); + std = 鍧楅棿娓呮櫚搴︾鏁e害 (灞閮ㄧ硦 -> 澶)銆""" + h, w = gray.shape[:2] + vals = [] + for i in range(grid): + for j in range(grid): + blk = gray[i * h // grid:(i + 1) * h // grid, j * w // grid:(j + 1) * w // grid] + if blk.size == 0: + continue + # 鍙畻"鏈夋枃瀛/杈圭紭缁撴瀯"鐨勫潡銆傚叧閿: 鐢ㄣ愯竟缂樺瘑搴︺戣岄潪 std 鍒ゆ湁鍐呭 鈥斺 + # std 鍙弽鏄犳槑鏆楄捣浼, 鐧界焊鐨勬姌鐥/闃村奖/娓愬彉 std 涔 >8, 浼氳璇綋"鍐呭鍧", + # 鑰岀櫧绾告媺鏅媺鏂瀬浣(~1) -> 鎶 worst_block 鎷夊埌鍋囨ф瀬浣 -> 娓呮櫚鏍囩琚鍒ょ硦銆 + # 鏂囧瓧鍧楁湁澶ч噺杈圭紭, 鐧界焊鍏夊奖鍧楀嚑涔庢棤杈圭紭 -> 鐢 Canny 杈圭紭鍗犳瘮鍖哄垎銆 + edges = cv2.Canny(blk, 50, 150) + edge_frac = float((edges > 0).mean()) + if edge_frac < 0.006: # 杈圭紭澶皯 = 鏃犳枃瀛楃粨鏋(鐧界焊/鑳屾櫙) -> 璺宠繃 + continue + vals.append(float(cv2.Laplacian(blk, cv2.CV_64F).var())) + if not vals: + return 0.0, 0.0 + worst = min(vals) + std = float(np.std(vals)) if len(vals) > 1 else 0.0 + return worst, std + + +def _local_sharp(gray, grid: int = 4) -> float: + """N脳N 鍒嗗潡, 鍙栨渶娓呮櫚鍧楃殑鎷夋櫘鎷夋柉鏂瑰樊銆""" + h, w = gray.shape[:2] + best = 0.0 + for i in range(grid): + for j in range(grid): + blk = gray[i * h // grid:(i + 1) * h // grid, j * w // grid:(j + 1) * w // grid] + if blk.size: + best = max(best, float(cv2.Laplacian(blk, cv2.CV_64F).var())) + return best + + +def _text_sharp(gray): + """MSER 鎵剧被鏂囧瓧鍊欓夊尯, 鍦ㄥ欓夊尯涓婄畻娓呮櫚搴︺傝繑鍥 (娓呮櫚搴, 闈㈢Н鍗犳瘮)銆 + 鎵句笉鍒 -> (0,0)銆傚崰姣=0 鎻愮ず'娌℃壘鍒板瓧'(鍙兘绯/鏃犲瓧), 涓 local_sharp 缁勫悎鍒よ銆""" + try: + mser = cv2.MSER_create(delta=5, min_area=60, max_area=14400) + regions, _ = mser.detectRegions(gray) + except Exception: + return 0.0, 0.0 + if not regions: + return 0.0, 0.0 + mask = np.zeros(gray.shape[:2], np.uint8) + for pts in regions: + x, y, bw, bh = cv2.boundingRect(pts.reshape(-1, 1, 2)) + ar = bw / max(bh, 1) + if 0.1 < ar < 10 and bw < gray.shape[1] * 0.9: + cv2.fillPoly(mask, [cv2.convexHull(pts.reshape(-1, 1, 2))], 255) + area_ratio = float((mask > 0).mean()) + if area_ratio < 0.001: + return 0.0, area_ratio + lap = cv2.Laplacian(gray, cv2.CV_64F) + return float(lap[mask > 0].var()), area_ratio + + +# ============================================================ +# 鍚庡彴鎶撳抚绾跨▼: 鐙崰 TCP, 鎸佺画鎶撴渶鏂板抚 (渚 /frame 鍙栫敤) +# 鍓嶇鎸夎嚜宸辩殑鍒锋柊鐜囨潵鍙, 鍚庣鍙淮鎶"鏈鏂颁竴甯+鎸囨爣"銆 +# ============================================================ +class Recorder: + """璁板綍涓娈典竴娈电殑鎸囨爣搴忓垪銆傛瘡娈靛甫涓涓 label(瑙﹀彂鏉ユ簮, 濡 af / res:UXGA / manual), + 姣忓抚涓琛, 鍔ㄦ佸垪(璺熼殢 frame_metrics 鐨 key)銆傚瓨纾佺洏 + 渚涘墠绔笅杞姐 + + 涓ょ瑙﹀彂: (a) 鐐规帶鍒舵寜閽嚜鍔ㄥ綍 duration 绉; (b) 鎵嬪姩寮濮/鍋滄銆 + 璁板綍鎸傚湪鍚庣鎶撳抚绾跨▼涓(姣忔姄涓甯ц涓琛), 涓庡墠绔樉绀哄埛鏂扮巼瑙h, 瀵嗗害=鐪熷疄鎶撳抚鐜囥""" + + def __init__(self, out_dir: Path): + self.out_dir = Path(out_dir) + self.out_dir.mkdir(parents=True, exist_ok=True) + self._lock = threading.Lock() + self._rows: list[dict] = [] # 鎵鏈夊凡璁板綍鐨勮 (璺ㄦ, 鍏ㄩ噺) + self._active = False + self._seg_label = "" + self._seg_start_mono = 0.0 + self._seg_deadline = 0.0 # >0 琛ㄧず瀹氭椂褰; 0 琛ㄧず鎵嬪姩褰(鏃犻檺鐩村埌 stop) + self._seg_index = 0 + self._columns: list[str] = [] # 鍔ㄦ佸垪, 棣栨璁板綍鏃舵寜 metrics key 纭畾 + + def start_segment(self, label: str, duration_s: float = 0.0): + """duration_s>0: 瀹氭椂褰; =0: 鎵嬪姩褰(鐩村埌 stop_segment)銆""" + with self._lock: + self._active = True + self._seg_index += 1 + self._seg_label = label + self._seg_start_mono = time.monotonic() + self._seg_deadline = (self._seg_start_mono + duration_s) if duration_s > 0 else 0.0 + + def stop_segment(self): + with self._lock: + self._active = False + + def feed(self, metrics: dict, grab_ms: float, res: str, rotate: int): + """鎶撳抚绾跨▼姣忓抚璋冧竴娆°備粎鍦 active 涓旀湭瓒呮椂鏃惰褰曘""" + with self._lock: + if not self._active: + return + now = time.monotonic() + if self._seg_deadline > 0 and now >= self._seg_deadline: + self._active = False + return + if not self._columns: + # 棣栨: 鍥哄畾鍓嶇紑鍒 + metrics 鐨勫姩鎬佸垪 + self._columns = (["seg", "label", "t_ms", "grab_ms", "res", "rotate"] + + list(metrics.keys())) + row = { + "seg": self._seg_index, + "label": self._seg_label, + "t_ms": round((now - self._seg_start_mono) * 1000, 1), + "grab_ms": round(grab_ms, 1), + "res": res, + "rotate": rotate, + } + row.update(metrics) + self._rows.append(row) + + def status(self) -> dict: + with self._lock: + remain = 0.0 + if self._active and self._seg_deadline > 0: + remain = max(0.0, self._seg_deadline - time.monotonic()) + return { + "active": self._active, + "label": self._seg_label, + "remain_s": round(remain, 1), + "n_rows": len(self._rows), + "n_segments": self._seg_index, + } + + def to_csv(self) -> str: + import csv, io + with self._lock: + rows = list(self._rows) + cols = list(self._columns) if self._columns else ["seg"] + # 鍔ㄦ佸垪鍙兘鍥犳湭鏉ユ寚鏍囧鍒犺屼笉榻: 鐢ㄦ墍鏈夎 key 鐨勫苟闆嗚ˉ榻 + allcols = list(cols) + for r in rows: + for k in r: + if k not in allcols: + allcols.append(k) + buf = io.StringIO() + w = csv.DictWriter(buf, fieldnames=allcols, extrasaction="ignore") + w.writeheader() + for r in rows: + w.writerow(r) + return buf.getvalue() + + def save_disk(self) -> Optional[Path]: + csv_text = self.to_csv() + stamp = time.strftime("%Y%m%d_%H%M%S") + path = self.out_dir / f"tuning_{stamp}.csv" + try: + path.write_text(csv_text, encoding="utf-8-sig") # BOM 渚夸簬 Excel 鐩存帴寮 + return path + except Exception as e: + print(f"[REC] 瀛樼洏澶辫触: {e}") + return None + + def clear(self): + with self._lock: + self._rows.clear() + self._seg_index = 0 + self._columns = [] + self._active = False + + +class FrameGrabber: + def __init__(self, tcp: TCPImageClient, recorder: "Recorder", ctl_state: dict): + self.tcp = tcp + self.recorder = recorder + self.ctl_state = ctl_state # {"res":..., "rotate":...} 渚涜褰曟爣娉ㄥ綋鍓嶇姸鎬 + self._latest_jpg: Optional[bytes] = None + self._latest_metrics: dict = {} + self._latest_grab_ms: float = 0.0 + self._rotate_deg = 0 + self._lock = threading.Lock() + self._stop = threading.Event() + self._thread = threading.Thread(target=self._run, daemon=True) + + def set_rotation(self, deg: int): + with self._lock: + self._rotate_deg = deg % 360 + + def start(self): + self._thread.start() + + def _run(self): + while not self._stop.is_set(): + t0 = time.monotonic() + jpg = self.tcp.capture() + dt = (time.monotonic() - t0) * 1000 + if jpg: + with self._lock: + deg = self._rotate_deg + if deg: + jpg = rotate_jpeg(jpg, deg) # 鐪熻浆, 鎸囨爣鍩轰簬杞悗鍥 + mtr = frame_metrics(jpg) + with self._lock: + self._latest_jpg = jpg + self._latest_metrics = mtr + self._latest_grab_ms = dt + # 鍠傝褰曞櫒 (浠呭湪 active 鏃剁湡璁); 甯﹀綋鍓嶅垎杈ㄧ巼/鏃嬭浆鐘舵 + self.recorder.feed(mtr, dt, + res=self.ctl_state.get("res", "?"), + rotate=deg) + else: + time.sleep(0.05) + + def latest(self): + with self._lock: + return self._latest_jpg, dict(self._latest_metrics), self._latest_grab_ms + + def grab_batch(self, n: int, rotate_apply: bool = True): + """杩炴姄 n 甯 (澶嶇敤 TCP 杩炴帴, 涓茶)銆傝繑鍥 (frames_list, grab_ms_list)銆 + 瀵瑰簲鐪熷疄閾捐矾 ASR 鍚'鎶 N 寮'閭d竴姝ャ""" + frames, lats = [], [] + with self._lock: + deg = self._rotate_deg + for _ in range(n): + t0 = time.monotonic() + jpg = self.tcp.capture() + dt = (time.monotonic() - t0) * 1000 + if jpg: + if rotate_apply and deg: + jpg = rotate_jpeg(jpg, deg) + frames.append(jpg) + lats.append(dt) + return frames, lats + + def stop(self): + self._stop.set() + self._thread.join(timeout=2.0) + self.tcp.close() + + +# ============================================================ +# Web 鏈嶅姟 +# ============================================================ +PAGE = r""" + + + + +鐩告満璋冨弬鍙 + + + +
+
+ +
+
Laplacian variance 路 娓呮櫚搴
+
+
+
+
+ +
绛夊緟鐢婚潰鈥
+
+
+
+
婕忔枟閫夊嚭鐨 BEST 甯 (鍠傜粰妯″瀷鐨勯偅寮)
+ +
+
+ + + + + + +""" + + +class PipelineStats: + """婕忔枟杩愯缁熻: 鎺ュ彈/鎷掔粷璁℃暟(鎸夊師鍥)+ best甯у瓨鐩 + 鏈杩 best 缂撳瓨銆 + 鎷掔粷鐜囨槸瀹夊叏鎸囨爣鐨勯洀褰: 璇ユ嫆鐨勬嫆浜嗗灏戙佸悇浠涔堝師鍥犮""" + + def __init__(self, out_dir: Path): + self.out_dir = Path(out_dir) + self.out_dir.mkdir(parents=True, exist_ok=True) + self._lock = threading.Lock() + self.n_total = 0 + self.n_accepted = 0 + self.reasons: dict = {} # reason -> count + self.last_best: Optional[bytes] = None + self._batch_i = 0 + self._runs: list[dict] = [] # 姣忔鎶撴媿涓琛屾眹鎬 (渚涙爣瀹氱敤) + + def add_run_row(self, scene: str, res, focused: bool, best_path: str = "", geom: dict = None): + """姣忔鎶撴媿杩藉姞涓琛屾眹鎬, 甯︾敤鎴锋爣娉ㄧ殑鍦烘櫙鏍囩 + best甯ц矾寰勩 + best_path: 渚涗綘浜嬪悗浜哄伐鐪嬭繖寮 best 娓呬笉娓呮, 鍦 manual_label 鍒楁爣鐪熷 鈥斺 + 鐢ㄤ汉宸ュ垽鏂(鑰岄潪鎶撴媿鍓嶆剰鍥炬爣绛)褰 ground truth 鏍″噯闃堝, 缁曞紑鏍囩閿欎綅銆""" + per = res.per_frame or [] + max_q = max((p.get("quality", 0) for p in per), default=0) + max_fg = max((p.get("focus_gain", 0) for p in per), default=0) + # best 甯х殑瀹屾暣 CV 鎸囨爣 (渚涙爣瀹: 涓寮犳眹鎬昏〃灏辫兘瀵规瘮鍚勬寚鏍 vs ok/blur) + bm = {} + if 0 <= res.best_index < len(per): + bm = per[res.best_index] + best_sharp = bm.get("sharpness", 0) + t = res.timings or {} + import os as _os + row = { + "time": time.strftime("%H:%M:%S"), + "scene": scene, + "manual_label": "", # 鈫 浣犵湅瀹 best 甯у悗鎵嬪~: ok / blur (鐪熷) + "n_frames": len(per), + "accepted": int(res.accepted), + "need_focus": int(res.need_focus), + "focused": int(focused), + "reject_reason": res.reject_reason, + "best_index": res.best_index, + "best_sharp": round(best_sharp, 1), + "best_local": bm.get("local_sharp", ""), + "best_worst_block": bm.get("worst_block", ""), + "best_block_std": bm.get("block_std", ""), + "best_text_sharp": bm.get("text_sharp", ""), + "best_text_ratio": bm.get("text_ratio", ""), + "best_contrast": bm.get("contrast", ""), + "best_dark_ratio": bm.get("dark_ratio", ""), + "seg_stable": (res.components or {}).get("stable", ""), + "seg_sharp": (res.components or {}).get("sharp", ""), + "seg_content": (res.components or {}).get("content", ""), + "seg_light": (res.components or {}).get("light", ""), + "max_quality": round(max_q, 3), + "max_focus_gain": round(max_fg, 3), + "grab_batch_ms": round(t.get("grab_batch_ms", 0), 1), + "funnel_ms": round(t.get("level1_metrics_ms", 0) + + t.get("level2_score_ms", 0) + + t.get("level3_decide_ms", 0), 1), + "total_ms": round(t.get("total_ms", 0), 1), + "best_file": _os.path.basename(best_path) if best_path else "", + "best_path": best_path, + # 鏂瑰悜鍒嗙被鍣ㄥ師濮嬭緭鍑(瀹氫綅鏂瑰悜涓嶅噯: raw鏍囩 vs 浣犲疄闄呮憜鏀) + "orient_raw": (geom or {}).get("orient_raw", ""), + "orient_conf": (geom or {}).get("orient_conf", ""), + "orient_state": (geom or {}).get("orient_state", ""), + "orient_hint": (geom or {}).get("orient_hint", "") or "", + } + with self._lock: + self._runs.append(row) + + def runs_csv(self) -> str: + import csv, io + with self._lock: + runs = list(self._runs) + cols = ["time", "scene", "manual_label", "n_frames", "accepted", "need_focus", + "focused", "reject_reason", "best_index", "best_sharp", + "best_local", "best_worst_block", "best_block_std", + "best_text_sharp", "best_text_ratio", "best_contrast", "best_dark_ratio", + "seg_stable", "seg_sharp", "seg_content", "seg_light", + "max_quality", "max_focus_gain", "grab_batch_ms", "funnel_ms", "total_ms", + "orient_raw", "orient_conf", "orient_state", "orient_hint", + "best_file", "best_path"] + # 鏈潵鏂板瓧娈佃嚜鍔ㄥ苟鍏 + for r in runs: + for k in r: + if k not in cols: + cols.append(k) + buf = io.StringIO() + w = csv.DictWriter(buf, fieldnames=cols, extrasaction="ignore") + w.writeheader() + for r in runs: + w.writerow(r) + return buf.getvalue() + + def save_runs_csv(self) -> Optional[Path]: + stamp = time.strftime("%Y%m%d_%H%M%S") + path = self.out_dir / f"pipeline_runs_{stamp}.csv" + try: + path.write_text(self.runs_csv(), encoding="utf-8-sig") + return path + except Exception as e: + print(f"[PSTATS] runs csv 瀛樼洏澶辫触: {e}") + return None + + def record(self, accepted: bool, reason: str): + with self._lock: + self.n_total += 1 + if accepted: + self.n_accepted += 1 + else: + self.reasons[reason] = self.reasons.get(reason, 0) + 1 + + def set_last_best(self, jpg: bytes): + with self._lock: + self.last_best = jpg + + def save_batch(self, frames: list[bytes], best_index: int, res): + """瀛樻暣鎵 N 甯 + 鏍囧嚭 best, 渚涗汉宸ユ牳瀵广傝繑鍥 (鐩綍, best甯ц矾寰)銆""" + with self._lock: + self._batch_i += 1 + bi = self._batch_i + d = self.out_dir / f"batch_{time.strftime('%H%M%S')}_{bi:03d}" + best_path = "" + try: + d.mkdir(parents=True, exist_ok=True) + for i, jpg in enumerate(frames): + tag = "_BEST" if i == best_index else "" + sharp = res.per_frame[i]["sharpness"] if i < len(res.per_frame) else 0 + fp = d / f"f{i}_sharp{sharp:.0f}{tag}.jpg" + fp.write_bytes(jpg) + if i == best_index: + best_path = str(fp) + import json + (d / "funnel.json").write_text( + json.dumps(res.to_dict(), ensure_ascii=False, indent=2), encoding="utf-8") + return str(d), best_path + except Exception as e: + print(f"[PSTATS] save_batch 澶辫触: {e}") + return "", "" + + def summary(self) -> dict: + with self._lock: + rej = self.n_total - self.n_accepted + return { + "total": self.n_total, + "accepted": self.n_accepted, + "rejected": rej, + "accept_rate": round(self.n_accepted / self.n_total, 3) if self.n_total else 0, + "reject_reasons": dict(self.reasons), + } + + +def make_app(grabber: FrameGrabber, ctl: CamControl, + recorder: Recorder, ctl_state: dict, default_dur: float, + funnel_cfg: "FunnelConfig", pstats: "PipelineStats") -> web.Application: + app = web.Application() + + async def h_index(request): + return web.Response(text=PAGE, content_type="text/html") + + async def h_frame(request): + jpg, m, grab_ms = grabber.latest() + if not jpg: + return web.Response(status=503, text="no frame") + headers = { + "X-Sharpness": str(m.get("sharpness", 0)), + "X-Brightness": str(m.get("brightness", 0)), + "X-Kb": str(m.get("kb", 0)), + "X-Grab-Ms": str(round(grab_ms, 0)), + "Cache-Control": "no-store", + } + return web.Response(body=jpg, content_type="image/jpeg", headers=headers) + + async def h_ctl(request): + op = request.query.get("op", "") + val = request.query.get("val", "") + dur = float(request.query.get("dur", default_dur) or default_dur) + loop = asyncio.get_event_loop() + label = None + if op == "res": + ok = await loop.run_in_executor(None, ctl.set_resolution, val) + ctl_state["res"] = val + label = f"res:{val}" + elif op == "af": + ok = await loop.run_in_executor(None, ctl.trigger_af) + label = "af" + elif op == "quality": + ok = await loop.run_in_executor(None, ctl.set_quality, int(val or 10)) + label = f"quality:{val}" + elif op == "rotate": + grabber.set_rotation(int(val or 0)) + ctl_state["rotate"] = int(val or 0) + ok = True + label = f"rotate:{val}" + else: + return web.json_response({"ok": False, "err": "unknown op"}, status=400) + # 鎺у埗鎿嶄綔鑷姩寮涓娈靛畾鏃惰褰 (鐪嬭鎿嶄綔鐨勬晥鏋滄洸绾) + if label and dur > 0: + recorder.start_segment(label, dur) + return web.json_response({"ok": bool(ok), "recording": label, "dur": dur}) + + # ---- 璁板綍鎺у埗 ---- + async def h_rec_start(request): + label = request.query.get("label", "manual") + dur = float(request.query.get("dur", 0) or 0) # 0=鎵嬪姩(鐩村埌 stop) + recorder.start_segment(label, dur) + return web.json_response({"ok": True, **recorder.status()}) + + async def h_rec_stop(request): + recorder.stop_segment() + return web.json_response({"ok": True, **recorder.status()}) + + async def h_rec_status(request): + return web.json_response(recorder.status()) + + async def h_rec_save(request): + path = recorder.save_disk() + return web.json_response({"ok": path is not None, + "path": str(path) if path else None, + **recorder.status()}) + + async def h_rec_download(request): + csv_text = recorder.to_csv() + recorder.save_disk() # 涓嬭浇鍚屾椂涔熷瓨鐩樹竴浠, 鍙屼繚闄 + return web.Response( + body=csv_text.encode("utf-8-sig"), + headers={"Content-Type": "text/csv; charset=utf-8", + "Content-Disposition": "attachment; filename=tuning.csv"}) + + async def h_rec_clear(request): + recorder.clear() + return web.json_response({"ok": True, **recorder.status()}) + + # ---- 褰㈡2: 鎶撴壒 -> 璇勫垎婕忔枟 -> (绋充絾绯婂垯瀵圭劍鍐嶉噰) -> best/鎷掔粷 ---- + async def h_pipeline_run(request): + n = int(request.query.get("n", 5) or 5) + focus_n = int(request.query.get("focus_n", 3) or 3) # 瀵圭劍鍚庤ˉ閲囧嚑甯 + af_settle_ms = int(request.query.get("af_settle_ms", 120) or 120) # 瀵圭劍鐢熸晥绛夊緟(ms) + # OV5640 鍗曟瀵圭劍(0x3022=0x03)閿佸畾閫氬父鍦▇100ms閲忕骇(鐩爣杩戠劍鏃舵洿蹇), 闈炲嚑鐧緈s銆 + # 榛樿120ms涓哄垵鍊, 瀹炴祴鐢 ?af_settle_ms=N 鎵弿鎵炬渶鐭湁鏁堢瓑寰呫 + scene = request.query.get("scene", "") # 鐢ㄦ埛鏍囨敞鍦烘櫙 + loop = asyncio.get_event_loop() + + def _work(): + t0 = time.monotonic() + frames, grab_lats = grabber.grab_batch(n) + grab_ms = (time.monotonic() - t0) * 1000 + if not frames: + return {"ok": False, "err": "no_frames"}, None, None + + res = run_funnel(frames, funnel_cfg) + focused = False + af_on = ctl_state.get("af_enabled", True) + # "鑷姩瀵圭劍"寮鍏 af_enabled 鐨勮涔(鐢ㄦ埛璁捐): + # ON(榛樿): 鑷富鍒ゆ柇+鎸夐渶瑙﹀彂瀵圭劍(璺嚎A, 姝e父浣跨敤)銆 + # OFF: 涓嶈嚜鍔ㄨЕ鍙戝鐒, 鍙繚鐣 res.need_focus 鐨勫垽鏂粨鏋 鈥斺 + # 鐢ㄤ簬褰曞埗瀹為獙鏃堕獙璇"鍒ゆ嵁璇磋瀵圭劍"杩欎釜鍒ゆ柇鏈韩瀵逛笉瀵, 涓嶈瀵圭劍鍔ㄤ綔骞叉壈銆 + # (鎵嬪姩"瑙﹀彂鍗曟AF"鎸夐挳璧 /op?op=af, 涓嶅彈姝ゅ紑鍏抽檺鍒, 浠讳綍鏃跺欏彲鎵嬪姩瑙﹀彂銆) + # 鍥轰欢渚ч』 g_auto_af=false, 鍚﹀垯鍥轰欢姣忕纭鐒︿細鏋剁┖姝ゅ紑鍏炽 + if res.need_focus and af_on: + tf = time.monotonic() + ctl.trigger_af() # 鍐 /reg 0x3022=0x03 鍗曟瀵圭劍 + # 绛夊鐒︾敓鏁堝啀鎶 鈥斺 trigger_af 鍚庨┈杈剧墿鐞嗗鐒﹂渶鏃堕棿(OV5640鍗曟瀵圭劍~100ms閲忕骇), + # 绔嬪埢 grab_batch 浼氭姄鍒"瀵圭劍杩囩▼涓"鐨勭硦甯с俛f_settle_ms 鍙皟, 瀹炴祴鏈鐭湁鏁堢瓑寰呫 + if af_settle_ms > 0: + time.sleep(af_settle_ms / 1000.0) # _work 鍦 executor 绾跨▼, sleep 涓嶉樆濉炰簨浠跺惊鐜 + extra, extra_lats = grabber.grab_batch(focus_n) + res.timings["focus_af_regrab_ms"] = (time.monotonic() - tf) * 1000 + if extra: + frames = frames + extra + grab_lats = grab_lats + extra_lats + res = run_funnel(frames, funnel_cfg) + focused = True + + res.timings["grab_batch_ms"] = grab_ms + res.timings["grab_per_frame_ms"] = [round(x, 1) for x in grab_lats] + res.timings["focused"] = 1 if focused else 0 + + infer_text = None + # 鍙鎬у甯т竴鑷存: 鏈獥鍙 accepted 鍙槸"鍊欓夐佷笅娓"; 澶嶇敤鏂瑰悜閭e璺ㄧ獥鍙g疮绉, + # 杩炵画绐楀彛澶氭暟閮藉欓夐, 鎵嶇湡姝i佷笅娓 鈥斺 婊ゆ帀"鍗曠獥鍙e伓鐒跺彲璇"(鏅冨姩涓伓鎶撴竻鏅板抚)銆 + # 涓ラ噸鏅冨姩(severe_shake)宸插湪 run_funnel 鍗曠獥鍙e嵆鎷, 涓嶈繘杩欓噷, 涓嶅彈涓鑷存у奖鍝嶃 + # 绾疌V鍒ゅ畾绱Н(涓嶅杺涓嬫父), 涓庢柟鍚戜竴鑷存у悓鏈哄埗銆侼=3 澶氭暟銆 + rhist = ctl_state.setdefault("readable_hist", []) + rhist.append(1 if res.accepted else 0) + del rhist[:-3] # 鍙暀鏈杩3绐楀彛 + readable_consistent = (len(rhist) >= 3 and sum(rhist) > 3 / 2) # 3绐楀彛澶氭暟 + res_dict_extra = {"readable_hist": list(rhist), + "readable_consistent": readable_consistent} + if res.accepted and res.best_jpg is not None and readable_consistent: + to = time.monotonic() + oriented, geom_diag = process_orientation(res.best_jpg) + res.timings["orientation_ms"] = (time.monotonic() - to) * 1000 + # 澶氬抚涓鑷存(鏂瑰悜): 绱Н鏈杩戣嫢骞叉鎶撴媿鐨"鐤戜技鍊掔疆"鏍囧織, 杩炵画/澶氭暟鎵嶅脊"璇风‘璁ゆ鍙"銆 + # 绾疌V鍒ゆ嵁鐨勫甯х疮绉(涓嶅杺VLM/OCR), 婊ゅ崟甯у櫔澹, 閬垮厤姝e父鍥惧伓鍙戣鎶ャ + if geom_diag and geom_diag.get("ok"): + hist = ctl_state.setdefault("flip_hist", []) + hist.append(1 if geom_diag.get("flipped_frame") else 0) + del hist[:-5] # 鍙暀鏈杩5娆 + # 5娆¢噷鈮3娆$枒浼煎掔疆, 涓旀湰娆′篃鐤戜技 -> 鎵嶅脊纭(澶氭暟涓鑷) + if len(hist) >= 3 and sum(hist) >= 3 and geom_diag.get("flipped_frame"): + if not geom_diag.get("orient_hint"): + geom_diag["orient_hint"] = "璇风‘璁よ嵂鐩掓鍙" + geom_diag["flip_consistent"] = True + geom_diag["flip_hist"] = list(hist) + ti = time.monotonic() + infer_text = infer(oriented) + res.timings["infer_ms"] = (time.monotonic() - ti) * 1000 + pstats.record(accepted=True, reason="") + saved_dir, best_path = pstats.save_batch(frames, res.best_index, res) + res.timings["total_ms"] = (time.monotonic() - t0) * 1000 + rd = res.to_dict(); rd.update(res_dict_extra) + return ({"ok": True, "result": rd, "focused": focused, + "infer": infer_text, "geom": geom_diag, "saved_dir": saved_dir}, + res.best_jpg, (res, focused, best_path)) + elif res.accepted and not readable_consistent: + # 鏈獥鍙g湅鐫鍙, 浣嗚繛缁竴鑷存ф湭杈(鍙兘鍋剁劧/鍒氬紑濮) -> 鏆備笉閫佷笅娓, 绛夌‘璁 + pstats.record(accepted=False, reason="await_consistency") + res.timings["total_ms"] = (time.monotonic() - t0) * 1000 + rd = res.to_dict(); rd.update(res_dict_extra) + rd["reject_reason"] = "await_consistency" + return ({"ok": True, "result": rd, "focused": focused, + "infer": None, "await_consistency": True}, + res.best_jpg, (res, focused, None)) + else: + # 鑻ュ鐒﹀悗浠 need_focus, 褰掍负鏈缁堟嫆缁(瀵圭劍涔熸病鏁戝洖鏉) + reason = res.reject_reason if res.reject_reason != "need_focus" else "all_blurry" + pstats.record(accepted=False, reason=reason) + res.reject_reason = reason + res.timings["total_ms"] = (time.monotonic() - t0) * 1000 + # 鎷掔粷鏃朵篃瀛樿瘉 + 鏄剧ず鏈鎵规閲屾渶娓呮櫚鐨勪竴甯(鑰岄潪鐣欐棫鍥), + # 璁╀綘鑳界湅鍒"绯荤粺杩欐闈㈠鐨勫埌搴曟槸浠涔堢敾闈", 澶嶇洏鍒ゅ畾鍐や笉鍐ゃ + rej_best = res.best_index if 0 <= res.best_index < len(frames) else 0 + saved_dir, best_path = pstats.save_batch(frames, rej_best, res) + show_jpg = frames[rej_best] if frames else None + return ({"ok": True, "result": res.to_dict(), "focused": focused, + "infer": None, "geom": None, "rejected_shown": True, + "saved_dir": saved_dir}, show_jpg, (res, focused, best_path)) + + result = await loop.run_in_executor(None, _work) + if len(result) == 3: + payload, best_jpg, run_info = result + else: + payload, best_jpg, run_info = result[0], result[1], None + if best_jpg is not None: + pstats.set_last_best(best_jpg) + if run_info is not None: + res_obj, focused, best_path = run_info + pstats.add_run_row(scene, res_obj, focused, best_path, payload.get("geom")) + return web.json_response({**payload, "stats": pstats.summary()}) + + async def h_pipeline_best(request): + jpg = pstats.last_best + if not jpg: + return web.Response(status=404) + return web.Response(body=jpg, content_type="image/jpeg", + headers={"Cache-Control": "no-store"}) + + async def h_pipeline_stats(request): + return web.json_response(pstats.summary()) + + async def h_pipeline_download(request): + csv_text = pstats.runs_csv() + pstats.save_runs_csv() # 鍚屾椂瀛樼洏 + return web.Response( + body=csv_text.encode("utf-8-sig"), + headers={"Content-Type": "text/csv; charset=utf-8", + "Content-Disposition": "attachment; filename=pipeline_runs.csv"}) + + # ---- 褰曞埗涓娈电函鍘熷甯ф祦(鎸夌, 涓嶆幒瀵圭劍), 渚 rerun 鐪嬭繃绋嬪垽瀹(闂1) ---- + async def h_record(request): + name = request.query.get("name", "").strip() or time.strftime("clip_%H%M%S") + dur = float(request.query.get("dur", 8) or 8) + dur = max(2.0, min(30.0, dur)) + safe = "".join(ch for ch in name if ch.isalnum() or ch in "_-") + clip_dir = pstats.out_dir.parent / "clips" / safe + loop = asyncio.get_event_loop() + try: + rec = await loop.run_in_executor( + None, record_stream, grabber, clip_dir, dur) + except Exception as e: + return web.json_response({"ok": False, "err": str(e)}, status=500) + return web.json_response({ + "ok": True, "dir": str(rec.dir), "name": safe, + "n_frames": len(rec.frames), "duration_s": dur, + }) + + app.router.add_get("/", h_index) + app.router.add_get("/frame", h_frame) + app.router.add_get("/ctl", h_ctl) + app.router.add_get("/rec/start", h_rec_start) + app.router.add_get("/rec/stop", h_rec_stop) + app.router.add_get("/rec/status", h_rec_status) + app.router.add_get("/rec/save", h_rec_save) + app.router.add_get("/rec/download", h_rec_download) + app.router.add_get("/rec/clear", h_rec_clear) + app.router.add_get("/pipeline/run", h_pipeline_run) + app.router.add_get("/pipeline/best", h_pipeline_best) + app.router.add_get("/pipeline/stats", h_pipeline_stats) + async def h_af_toggle(request): + val = request.query.get("on", "") + if val in ("1", "true", "on"): + ctl_state["af_enabled"] = True + elif val in ("0", "false", "off"): + ctl_state["af_enabled"] = False + return web.json_response({"ok": True, "af_enabled": ctl_state.get("af_enabled", True)}) + + app.router.add_get("/pipeline/download", h_pipeline_download) + app.router.add_get("/record", h_record) + app.router.add_get("/af_toggle", h_af_toggle) + return app + + +# ============================================================ +# 褰曞埗 + rerun (鍙楁帶瀵规瘮鍩虹) +# 褰曞埗: 閲囦竴娈(鏈鐒) -> 瑙﹀彂鐪烝F -> 鍐嶉噰涓娈(瀵圭劍鍚), 涓ゆ閮藉瓨, 鏍囧鐒︾偣銆 +# rerun: 浠庡綍鍒跺洖鏀惧抚鍠 run_funnel, 鍚屼竴娈靛弽澶嶆壂鍙傛暟, 杈撳叆鍥哄畾=鍙楁帶瀵规瘮銆 +# 瑙f硶涓: 瀵圭劍鏁堟灉涔熷綍杩涘幓(瀵圭劍鍚庣殑甯х湡瀹炲瓨鍦), 鏁 C 鍑哄彛鐨勫鐒﹀彲鍙楁帶澶嶇幇銆 +# ============================================================ +class Recording: + """涓娈靛綍鍒: 鍘熷甯у簭鍒 + 姣忓抚鏃堕棿鎴 + 瀵圭劍瑙﹀彂鐐规爣璁般""" + + def __init__(self, clip_dir: Path): + self.dir = Path(clip_dir) + self.frames: list[bytes] = [] + self.timestamps: list[float] = [] + self.af_index = -1 # 瀵圭劍瑙﹀彂鐐: 姝ょ储寮曞強涔嬪悗鐨勫抚鏄"瀵圭劍鍚" + self.meta = {} + + def save(self): + self.dir.mkdir(parents=True, exist_ok=True) + for i, jpg in enumerate(self.frames): + (self.dir / f"frame_{i:04d}.jpg").write_bytes(jpg) + import json + meta = { + "n_frames": len(self.frames), + "timestamps": self.timestamps, + "af_index": self.af_index, # -1=鏃犲鐒︾偣 + **self.meta, + } + (self.dir / "meta.json").write_text( + json.dumps(meta, ensure_ascii=False, indent=2), encoding="utf-8") + return self.dir + + @classmethod + def load(cls, clip_dir): + import json + d = Path(clip_dir) + rec = cls(d) + meta = json.loads((d / "meta.json").read_text(encoding="utf-8")) + rec.af_index = meta.get("af_index", -1) + rec.timestamps = meta.get("timestamps", []) + rec.meta = meta + n = meta.get("n_frames", 0) + for i in range(n): + fp = d / f"frame_{i:04d}.jpg" + if fp.exists(): + rec.frames.append(fp.read_bytes()) + return rec + + def pre_af_frames(self): + """瀵圭劍鍓嶇殑甯 (rerun 鏃跺厛鐢ㄨ繖浜涜窇婕忔枟)銆""" + if self.af_index < 0: + return self.frames + return self.frames[:self.af_index] + + def post_af_frames(self): + """瀵圭劍鍚庣殑甯 (rerun 鏃 C 鍑哄彛瑙﹀彂瀵圭劍鍚庡洖鏀捐繖浜)銆""" + if self.af_index < 0: + return [] + return self.frames[self.af_index:] + + +def record_stream(grabber: "FrameGrabber", clip_dir: Path, + duration_s: float = 8.0, gap_s: float = 0.05) -> Recording: + """鎸夌褰曚竴娈电函鍘熷甯ф祦 (涓嶆幒瀵圭劍)銆備笓涓洪棶棰1: rerun 鍥炴斁鐪嬫暣涓繃绋嬬殑 + 鍒ゅ畾搴忓垪, 楠岃瘉 CV 鍒ょ殑'璇ュ鐒'瀵逛笉瀵广俛f_index=-1 琛ㄧず鏃犲鐒︾偣銆""" + rec = Recording(clip_dir) + rec.af_index = -1 + t0 = time.monotonic() + while (time.monotonic() - t0) < duration_s: + jpg = grabber.tcp.capture() + if jpg: + rec.frames.append(jpg) + rec.timestamps.append(time.monotonic() - t0) + time.sleep(gap_s) + rec.meta["duration_s"] = duration_s + rec.meta["kind"] = "stream" # 鍖哄埆浜庡鐒﹀姣旂殑 clip + rec.save() + return rec + + +def rerun_stream(clip_dir, cfg: "FunnelConfig" = None, n: int = 5, stride: int = 3): + """瀵逛竴娈佃繛缁綍鍒, 婊戝姩绐楀彛璺戝垽瀹, 杈撳嚭'鏁翠釜杩囩▼鐨勫垽瀹氬簭鍒'銆 + 姣 stride 甯ф粦涓娆, 绐楀彛澶у皬 n銆傜湅鐗╁搧杩->鍋->鍑烘椂鍒ゅ畾(A/B/C)鎬庝箞鍙樸 + 瀵圭劍鍛戒护澶╃劧鍏抽棴(鍥炴斁姝诲浘), 鍗抽棶棰1: 楠岃瘉'璇ュ鐒'鐨勫垽鏂涓嶅銆 + + 浜у嚭(渚涗汉宸ユ牳瀵): 姣忕獥鍙e瓨 best 甯(鏂囦欢鍚嶅甫鍒ゅ畾) + 涓涓 CSV(姣忕獥鍙d竴琛, + 甯﹀垽瀹/鍚勫垎閲/best甯ц矾寰)銆備綘绛 reason=need_focus 鐨勮, 鐐瑰紑鍘熷浘鐪嬫槸鍚︾湡闇璋冪劍銆""" + import csv, json + cfg = cfg or FunnelConfig() + rec = Recording.load(clip_dir) + frames = rec.frames + out_dir = Path(clip_dir) / "rerun" + out_dir.mkdir(parents=True, exist_ok=True) + print(f"\n=== 杩囩▼鍒ゅ畾搴忓垪 {Path(clip_dir).name} " + f"({len(frames)}甯, ~{rec.meta.get('duration_s','?')}s, 绐楀彛N={n} 姝ラ暱={stride}) ===") + print(f"{'win':<5}{'t(s)':<7}{'鍒ゅ畾':<10}{'reason':<14}{'stable':<8}{'sharp':<8}{'content':<8}{'best鍥'}") + print("-" * 78) + rows = [] + i = 0 + win = 0 + flip_hist = [] # 璺ㄧ獥鍙e甯т竴鑷存: 绱Н"鐤戜技鍊掔疆"鏍囧織 + while i + n <= len(frames): + window = frames[i:i + n] + res = run_funnel(window, cfg) + t = rec.timestamps[i] if i < len(rec.timestamps) else i * 0.05 + comp = res.components or {} + if res.accepted: + verdict = "B_usable" + elif res.reject_reason == "need_focus": + verdict = "C_needfocus" + elif res.reject_reason == "unstable": + verdict = "A_moving" + else: + verdict = res.reject_reason + # 瀛 best 甯 (绐楀彛鍐 best_index; 鎷掔粷鏃朵篃瀛樺弬鑰冨抚) + bidx = res.best_index if 0 <= res.best_index < len(window) else 0 + best_name = f"w{win:03d}_t{t:.1f}_{verdict}.jpg" + (out_dir / best_name).write_bytes(window[bidx]) + # best 甯х殑瀹屾暣鎸囨爣 + bm = res.per_frame[bidx] if 0 <= bidx < len(res.per_frame) else {} + # === 鏂瑰悜鍒ゆ嵁 (鍙琚帴鍙楃殑 best 甯ц窇; 鎷掔粷鐨勫抚鏈濆悜鏃犳剰涔) === + orient_state = ""; ud_score = ""; side_ratio = ""; orient_hint = ""; flip_consistent = "" + if res.accepted: + _, gd = process_orientation(window[bidx]) + if gd.get("ok"): + orient_state = gd.get("orient_state", "") + ud_score = gd.get("ud_score", "") + side_ratio = gd.get("sideways_ratio", "") + orient_hint = gd.get("orient_hint") or "" + # 澶氬抚涓鑷存(鎸夌獥鍙i『搴忕疮绉, 涓 live 鍚岄昏緫): 5绐楀唴鈮3绐楃枒鍊掔疆涓旀湰绐椾篃鐤 -> 瑙﹀彂 + flip_hist.append(1 if gd.get("flipped_frame") else 0) + del flip_hist[:-5] + if len(flip_hist) >= 3 and sum(flip_hist) >= 3 and gd.get("flipped_frame"): + flip_consistent = 1 + if not orient_hint: + orient_hint = "璇风‘璁よ嵂鐩掓鍙" + else: + flip_consistent = 0 + print(f"{win:<5}{t:<7.1f}{verdict:<10}{res.reject_reason:<14}" + f"{comp.get('stable',0):<8.2f}{comp.get('sharp',0):<8.2f}{comp.get('content',0):<8.2f}{best_name}") + rows.append({ + "win": win, "t_s": round(t, 2), "verdict": verdict, + "reject_reason": res.reject_reason, "accepted": int(res.accepted), + "need_focus": int(res.need_focus), + "g_stable": comp.get("stable", ""), "g_sharp": comp.get("sharp", ""), + "g_content": comp.get("content", ""), "g_light": comp.get("light", ""), + "seg_flow": comp.get("flow", ""), + "best_sharp": bm.get("sharpness", ""), "best_local": bm.get("local_sharp", ""), + "best_worst_block": bm.get("worst_block", ""), "best_text_ratio": bm.get("text_ratio", ""), + "best_frame": best_name, + # 鏂瑰悜鍒ゆ嵁杈撳嚭 (渚涙牳瀵规柟鍚戝噯纭) + "orient_state": orient_state, "ud_score": ud_score, + "sideways_ratio": side_ratio, "flip_consistent": flip_consistent, + "orient_hint": orient_hint, + "manual_check": "", # 鈫 浣犵湅瀹 best 鍥惧悗濉: 璇ュ鐒︾殑鐪熻鍚? ok/wrong + "manual_orient": "", # 鈫 浣犲~杩欑獥 best 鐨勭湡瀹炴湞鍚: upright/sideways/flipped + }) + i += stride + win += 1 + # 瀛 CSV + cols = ["win", "t_s", "verdict", "reject_reason", "accepted", "need_focus", + "g_stable", "g_sharp", "g_content", "g_light", "seg_flow", + "best_sharp", "best_local", "best_worst_block", "best_text_ratio", + "best_frame", + "orient_state", "ud_score", "sideways_ratio", "flip_consistent", "orient_hint", + "manual_check", "manual_orient"] + csv_path = out_dir / "sequence.csv" + with open(csv_path, "w", encoding="utf-8-sig", newline="") as f: + w = csv.DictWriter(f, fieldnames=cols) + w.writeheader() + w.writerows(rows) + from collections import Counter + cnt = Counter(r["verdict"] for r in rows) + print(f"\n鍒ゅ畾鍒嗗竷: {dict(cnt)}") + print(f"CSV: {csv_path}") + print(f"best鍥: {out_dir}/ (绛 need_focus 鐨勭獥鍙, 鐪嬪搴 best鍥炬槸鍚︾湡闇璋冪劍)") + return rows + + +def record_clip(grabber: "FrameGrabber", ctl: "CamControl", clip_dir: Path, + pre_n: int = 8, post_n: int = 8, gap_s: float = 0.05) -> Recording: + """褰曚竴娈: 鎶 pre_n 甯(鏈鐒) -> 瑙﹀彂鐪烝F -> 鎶 post_n 甯(瀵圭劍鍚)銆 + 鍚鐒︾偣鏍囪, 渚 rerun 澶嶇幇 C 鍑哄彛瀵圭劍鏁堟灉 (瑙f硶涓)銆""" + rec = Recording(clip_dir) + t0 = time.monotonic() + # 鏈鐒︽ + for _ in range(pre_n): + jpg = grabber.tcp.capture() + if jpg: + rec.frames.append(jpg) + rec.timestamps.append(time.monotonic() - t0) + time.sleep(gap_s) + # 瑙﹀彂鐪 AF + rec.af_index = len(rec.frames) + ctl.trigger_af() + time.sleep(0.2) # 缁欏鐒︿竴鐐圭敓鏁堟椂闂 + # 瀵圭劍鍚庢 + for _ in range(post_n): + jpg = grabber.tcp.capture() + if jpg: + rec.frames.append(jpg) + rec.timestamps.append(time.monotonic() - t0) + time.sleep(gap_s) + rec.meta["pre_n"] = pre_n + rec.meta["post_n"] = post_n + rec.save() + return rec + + +def rerun_clip(rec: "Recording", cfg: "FunnelConfig", n: int, focus_n: int = 3) -> dict: + """瀵逛竴娈靛綍鍒惰窇婕忔枟: 鐢ㄥ墠 n 甯(瀵圭劍鍓); 鑻ュ垽 need_focus, 鐢ㄥ鐒﹀悗甯ч噸璇勩 + 杈撳叆鍥哄畾, 鍙彉 cfg/n/focus_n -> 鍙楁帶瀵规瘮銆傝繑鍥炲垽瀹 + 鏄惁鐢ㄤ簡瀵圭劍銆""" + pre = rec.pre_af_frames() + if len(pre) < n: + n = len(pre) + batch = pre[:n] + res = run_funnel(batch, cfg) + focused = False + if res.need_focus: + post = rec.post_af_frames() + if post: + extra = post[:focus_n] + res = run_funnel(batch + extra, cfg) # 鍚堝苟閲嶈瘎, best 搴旀潵鑷鐒﹀悗甯 + focused = True + return { + "accepted": res.accepted, + "reject_reason": res.reject_reason, + "need_focus": res.need_focus, + "focused": focused, + "best_index": res.best_index, + "components": res.components, + "n": n, "focus_n": focus_n, + } + + +def rerun_sweep(clip_dir, cfg: "FunnelConfig" = None, + n_values=(3, 5, 7), focus_values=(0, 3)): + """瀵逛竴娈靛綍鍒舵壂鍙傛暟, 鎵撳嵃缁撴灉琛 (鎵 N 鎷愮偣 / 瀵圭劍鏄惁鍊煎緱)銆 + 鍙楁帶瀵规瘮: 鍚屼竴娈靛抚, 鍙彉 N 鍜 focus_n銆""" + cfg = cfg or FunnelConfig() + rec = Recording.load(clip_dir) + print(f"\n=== rerun 鎵弬 {Path(clip_dir).name} " + f"({len(rec.frames)}甯, 瀵圭劍鐐笯{rec.af_index}) ===") + print(f"{'N':<5}{'focus_n':<9}{'鍒ゅ畾':<12}{'鐢ㄥ鐒':<8}{'best':<6}{'鍘熷洜'}") + print("-" * 50) + for n in n_values: + for fn in focus_values: + r = rerun_clip(rec, cfg, n, fn) + verdict = "鎺ュ彈" if r["accepted"] else "鎷掔粷" + print(f"{n:<5}{fn:<9}{verdict:<12}{'鏄' if r['focused'] else '鍚':<8}" + f"{r['best_index']:<6}{r['reject_reason']}") + return rec + + +def main(): + ap = argparse.ArgumentParser(description="ESP32 鐩告満娴佹按绾垮彴 (褰㈡2: 婕忔枟+褰曞埗+rerun)") + ap.add_argument("--ip", help="鐪奸暅 IP (live/record 妯″紡蹇呴渶)") + ap.add_argument("--tcp-port", type=int, default=5000) + ap.add_argument("--port", type=int, default=8080, help="缃戦〉绔彛") + ap.add_argument("--record-dir", default="./tuning_logs", help="CSV/鎵规 瀛樼洏鐩綍") + ap.add_argument("--duration", type=float, default=5.0, help="鐐规帶鍒舵寜閽悗鑷姩璁板綍绉掓暟") + ap.add_argument("--quality", type=int, default=4, help="鍚姩榛樿 JPEG 璐ㄩ噺(灏=娓呮櫚, 榛樿4鏈楂)") + # 褰曞埗 / rerun 妯″紡 + ap.add_argument("--record", metavar="CLIP_DIR", help="褰曚竴娈(鍚鐒︾偣)鍒版寚瀹氱洰褰, 鐒跺悗閫鍑") + ap.add_argument("--pre-n", type=int, default=8, help="褰曞埗: 瀵圭劍鍓嶅抚鏁") + ap.add_argument("--post-n", type=int, default=8, help="褰曞埗: 瀵圭劍鍚庡抚鏁") + ap.add_argument("--rerun", metavar="CLIP_DIR", help="瀵逛竴娈靛綍鍒舵壂鍙傛暟(N/focus_n), 鐒跺悗閫鍑") + args = ap.parse_args() + + # ---- rerun 妯″紡: 绾绾, 涓嶈繛鐪奸暅 ---- + if args.rerun: + rec = Recording.load(args.rerun) + if rec.meta.get("kind") == "stream" or rec.af_index < 0: + rerun_stream(args.rerun) # 杩炵画杩囩▼ -> 杩囩▼鍒ゅ畾搴忓垪(闂1) + else: + rerun_sweep(args.rerun) # 瀵圭劍瀵规瘮 clip -> 鎵弬鏁 + return + + # ---- record 妯″紡: 杩炵溂闀滃綍涓娈 ---- + if args.record: + if not args.ip: + print("record 妯″紡闇瑕 --ip"); return + tcp = TCPImageClient(args.ip, args.tcp_port) + grabber = FrameGrabber(tcp, Recorder(Path(args.record_dir)), {"res": "HD", "rotate": 0}) + grabber.start(); time.sleep(0.3) + ctl = CamControl(args.ip) + print(f"[REC] 褰曞埗涓: {args.pre_n}甯(鏈鐒) -> AF -> {args.post_n}甯(瀵圭劍鍚)...") + rec = record_clip(grabber, ctl, Path(args.record), args.pre_n, args.post_n) + print(f"[REC] 宸插瓨 {rec.dir} ({len(rec.frames)}甯, 瀵圭劍鐐笯{rec.af_index})") + print(f"[REC] rerun: python cam_pipeline.py --rerun {rec.dir}") + grabber.stop() + return + + # ---- live 妯″紡: 缃戦〉 ---- + if not args.ip: + print("live 妯″紡闇瑕 --ip"); return + recorder = Recorder(Path(args.record_dir)) + ctl_state = {"res": "HD", "rotate": 0, "af_enabled": True} + funnel_cfg = FunnelConfig() + pstats = PipelineStats(Path(args.record_dir) / "batches") + + tcp = TCPImageClient(args.ip, args.tcp_port) + grabber = FrameGrabber(tcp, recorder, ctl_state) + grabber.start() + ctl = CamControl(args.ip) + # 鍚姩榛樿: HD + 鏈楂樼敾璐(quality=4), 鐪佸緱姣忔杩涚綉椤垫墜璋 + ctl.set_resolution("HD"); ctl_state["res"] = "HD" + ctl.set_quality(args.quality) + time.sleep(0.15); ctl.trigger_af() # 鍒囧畬琛ヤ竴娆″鐒 + print(f"[PIPE] 鍚姩榛樿: HD + quality={args.quality} + 瀵圭劍") + + app = make_app(grabber, ctl, recorder, ctl_state, args.duration, funnel_cfg, pstats) + print(f"[PIPE] http://localhost:{args.port} (TCP {args.ip}:{args.tcp_port})") + print(f"[PIPE] 褰曞埗鍙楁帶瀵规瘮: python cam_pipeline.py --ip {args.ip} --record clips/case1") + print("[PIPE] 鐙崰 TCP: 璋冨弬鏃朵笉瑕佸悓鏃惰窇涓荤▼搴") + try: + web.run_app(app, host="0.0.0.0", port=args.port, print=None) + finally: + grabber.stop() + + +if __name__ == "__main__": + main() diff --git a/extensions/assistive_harness/phase_b/esp32_runtime.py b/extensions/assistive_harness/phase_b/esp32_runtime.py new file mode 100644 index 0000000..285eec2 --- /dev/null +++ b/extensions/assistive_harness/phase_b/esp32_runtime.py @@ -0,0 +1,1971 @@ +"""ESP32 Phase B Harness runtime. + +澶嶇敤 rokid_runtime 鐨勫叏閮 harness 楠ㄦ灦锛圚arnessClient / GatewaySessionManager / +GatewayDuplexSession / PCSpeaker / OutputGate / 鎺у埗閫昏緫锛夛紝鍙妸璁惧 I/O 浠 +"PC 璧 web server 绛 Rokid 鎺" 鎹㈡垚 "PC 涓诲姩杩 ESP32 鎷夐煶瑙嗛"銆 + +鏁版嵁鏂瑰悜瀵规瘮锛 + Rokid : 鐪奸暅(APK) --push--> PC 鐨 aiohttp server + ESP32 : PC --pull--> ESP32(CameraWebServer, /ws_audio_v2 + TCP:5000) + +姹囪仛鐐瑰畬鍏ㄤ竴鑷达細 + 闊抽 -> audio_queue(缁 gateway session) + harness.send_audio(闀滃儚缁 8021 ASR) + 鍥惧儚 -> latest_frame.set() + harness.send_frame(闀滃儚缁 8021) + AI 闊抽杈撳嚭 -> PCSpeaker +""" +from __future__ import annotations + +import argparse +import asyncio +import json +import logging +import signal +import ssl +import struct +import base64 +import os +import time +from pathlib import Path +from typing import Any, Optional + +import aiohttp +import numpy as np + +from .timing_probe import TimingProbe +from .funnel_gate import FunnelGate +from .session_recorder import SessionRecorder +from .recorder_live import LiveRecorder +try: + from .bridge_ui import WebUIServer +except Exception: + WebUIServer = None + +# 鈹鈹 澶嶇敤 rokid_runtime 鐨 harness 楠ㄦ灦锛堜竴涓瓧涓嶆敼锛夆攢鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹 +from .rokid_runtime import ( + RokidRuntimeConfig, + GatewaySessionManager, + GatewayDuplexSession, + HarnessClient, + PCSpeaker, + NullSpeaker, + OutputGate, + DropOldestAudioQueue, + AudioMirrorChunker, + LatestFrame, + RuntimeStats, + CtrlCExitWatchdog, + SkillRegistry, + default_skills_config, + pcm16le_to_float32, + apply_pcm16_gain, + now_ms, + SAMPLE_RATE_IN, +) + +LOG = logging.getLogger("assistive_harness.phase_b.esp32") + +# ESP32 闊抽鍖呭ご锛 = seq(4) ts_ms(4) n_samples(2) reserved(2) +# 娉ㄦ剰锛氱4瀛楁鏄 reserved(pad)锛屽浐浠剁湡涓㈠寘璁℃暟 g_ring_drops 鏈彂鍒 wire锛 +# PC 绔敤 seq 璺冲彉鎺ㄦ柇鐪熶涪鍖呫 +ESP32_PKT_HDR = 12 +# TCP 鍥惧儚甯уご magic +_TCP_IMG_MAGIC = 0x55AA55AA + +# 鍙ュ瓙杈圭晫鏍囩偣锛堜綔涓"淇濇姢鍗曚綅"鐨勮竟鐣岋級 +def _install_asr_tap(harness, live_rec) -> None: + """鎶 8021 鐨 asr.transcript 鎴笅鏉ヨ惤鐩橈紙璇勫垎鍙栨彁闂椂鍒 t0 鐢級銆 + + 涓轰粈涔堜笉鏄彟寮涓鏉¤繛鎺ワ細8021 姣忎釜 client_id 鏈夌嫭绔嬬殑 runtime 鍜 outbound 闃熷垪锛 + asr.transcript 鍙姇閫掔粰"閫侀煶棰戣繘鏉ョ殑閭d釜 client"锛堝嵆 esp32-phase-b锛夛紝 + 鍙﹀紑鐨 esp32-asr-tap 铏界劧鑳借繛涓婏紝浣嗘案杩滄敹涓嶅埌浠讳綍娑堟伅銆 + 鎵浠ュ彧鑳芥寕鍦ㄥ悓涓鏉 HarnessClient 涓娾斺攔okid_runtime 閲屽姞浜嗕釜鍙夌殑 on_message + 鏃佽矾锛堥粯璁 None锛屼笉褰卞搷鍘熻涓猴級锛岃繖閲岃祴鍊煎嵆鍙 + + 8021 鎺ㄧ殑瀛楁锛坰erver.py 148-154锛夛細 + {"type":"asr.transcript", "asr_event_id":..., "utterance":"杩欎笂闈㈠啓鐨勬槸浠涔", + "confidence":..., "final_at_ms":..., + "suppressed": true/false, "reason": "..."} 鈫 琚洖澹版姂鍒舵椂甯﹁繖涓や釜 + 娉ㄦ剰锛**琚 echo drop 鐨勪篃鐓ф牱鎺**锛堟姂鍒跺彂鐢熷湪 route/echo 涔嬪悗锛夛紝鎵浠ヨ繖閲屽叏閮借锛 + 鐢辫瘎鍒嗕晶鎸夐棶鍙ョ瓫锛屼笉鍦ㄨ繖閲岃繃婊ゃ + """ + if live_rec is None or harness is None: + return + + def _on_message(payload): + if not isinstance(payload, dict): + return + if payload.get("type") != "asr.transcript": + return + utt = str(payload.get("utterance") or "").strip() + if not utt: + return + try: + live_rec.log_asr(utt, payload.get("final_at_ms"), + payload.get("confidence")) + LOG.info("[ASR] %s%s", utt, + " (suppressed)" if payload.get("suppressed") else "") + except Exception as e: + LOG.warning("[ASR] 钀界洏澶辫触: %s", e) + + harness.on_message = _on_message + LOG.info("[ASR] transcript 鏃佽矾宸叉寕鍦 HarnessClient 涓") + + +class TurnPrinter: + """鎶婇 chunk 鐨 text 纰庣墖鑱氬悎鎴愭暣娈 turn锛堢Щ妞嶈嚜 demo_esp32_duplex_0703锛夈 + + -o 姣忎釜 chunk 鍙悙鍑犱釜瀛楋紝璇勫垎瑕佺殑鏄"杩欐鍥炵瓟璇翠簡浠涔"銆傝鍒欙細 + text 闈炵┖ 涓旓紙涓嶅湪 turn 鍐 鎴 listen/speak 鍙戠敓鍒囨崲锛夆啋 寮鏂 turn锛宼urn_idx +1 + end_of_turn 鈫 杩斿洖 (turn_idx, 鏁存鏂囨湰, is_listen) 骞剁粨鏉熸湰 turn + is_listen=True 鐨 turn 鏄 8021 ASR 璇嗗埆鍑虹殑鐢ㄦ埛璇磋瘽锛堢敤浜庡彇鎻愰棶缁撴潫鏃跺埢 t0锛夈 + """ + + def __init__(self) -> None: + self.turn_idx = 0 + self._in_turn = False + self._turn_is_listen = None + self._turn_text: list[str] = [] + + def _reset(self) -> None: + self._in_turn = False + self._turn_text.clear() + + def feed(self, is_listen: bool, end_of_turn: bool, text: str): + """杩斿洖 (turn_idx, 瀹屾暣鏂囨湰鎴"", is_listen)銆 + + 闄 end_of_turn 澶栵紝**listen/speak 鍒囨崲鏃朵篃浼氭妸涓婁竴娈靛悙鍑烘潵** 鈥斺 + 鍘熷疄鐜板湪鍒囨崲鏃剁洿鎺 _reset() 涓㈡帀锛屾ā鍨嬩竴璺祦寮忚銆乪nd_of_turn 杩熻繜涓嶆潵鏃 + 鏁存灏辨病浜嗭紙bare 缁勫疄娴 transcript 鍏ㄧ┖锛夈 + """ + flushed = None + if text: + if (not self._in_turn) or (self._turn_is_listen != is_listen): + if self._in_turn and self._turn_text: + flushed = (self.turn_idx, "".join(self._turn_text), + bool(self._turn_is_listen)) + self._reset() + self.turn_idx += 1 + self._in_turn = True + self._turn_is_listen = is_listen + self._turn_text.append(text) + if flushed is not None and not end_of_turn: + return flushed + if end_of_turn: + full_text = "".join(self._turn_text) + cur_idx = self.turn_idx + cur_listen = bool(self._turn_is_listen) + self._reset() + return cur_idx, full_text, cur_listen + return self.turn_idx, "", is_listen + + +_SENT_PUNCT = "锛屻傦紒锛,.!?" + + +def _emit_chunk_img(web_ui, jpeg: bytes, idx: int, img_sent: bool, age_ms: int = 0, + reason: str = ""): + """鎶婁竴甯у浘鎺ㄧ粰 bridge_ui 绗竴瑙嗚锛坱ype=chunk, img_b64锛夈傛棤瀹㈡埛绔椂闆惰礋鎷呫 + + img_sent=False 琛ㄧず杩欏抚娌¢佹ā鍨嬶紙琚紡鏂楁嫤涓嬫垨涓嶆槸 best锛夈傜涓瑙嗚鐓ф牱鏄剧ず锛 + 杩欐牱鐢婚潰鎵嶈繛缁佷篃鑳界湅瑙"琚嫤鐨勬槸浠涔堟牱鐨勫浘"鈥斺斿婕旂ず鈶e弽鑰屾洿鏈夎鏈嶅姏銆 + """ + if web_ui is None or not getattr(web_ui, "live_clients", None): + return + try: + b64 = base64.b64encode(jpeg).decode("ascii") if jpeg else None + import asyncio as _a + _a.create_task(web_ui.emit({ + "type": "chunk", + "idx": int(idx), + "img_b64": b64, + "img_sent": bool(img_sent), + "img_age_ms": int(age_ms), + "reject_reason": reason or "", + })) + except Exception: + pass + + +class SentenceTracker: + """璺熻釜妯″瀷杈撳嚭鐨勫彞瀛愯竟鐣岋紝渚涙紡鏂"鏍囩偣淇濇姢"绛栫暐鐢ㄣ + + handle_result 姣忎釜 chunk 璋 update()锛氭娴嬪彞鏈爣鐐癸紝绱鏍囩偣搴忓彿銆 + image_loop锛歳eject 鎸傝捣鏃 snapshot() 璁板綋鍓嶆爣鐐瑰簭鍙凤紱涔嬪悗 + boundary_reached(snap, speaker) 鍒ゆ柇"鑷寕璧峰悗鏄惁鍑虹幇浜嗘柊鍙ユ湯鏍囩偣锛 + 涓旇娈佃闊冲凡缁忔挱瀹岋紙speaker 闃熷垪鍩烘湰娓呯┖锛"銆 + + 鍏抽敭锛氭ā鍨嬫枃瀛楃敓鎴愯繙蹇簬璇煶鎾斁锛屾枃瀛楁爣鐐逛細鏃╀簬璇煶鍒拌揪銆傛墍浠"璇煶鎾埌 + 鏍囩偣"涓嶈兘闈犳枃瀛楁爣鐐规椂鍒绘垨澧欓挓浼扮畻锛岃岃鐪 speaker.pending_ms 鈥斺 闃熷垪閲岃繕 + 娌℃挱鐨勮闊抽檷鍒板緢浣庢椂锛屾墠璇存槑褰撳墠杩欐锛堝惈鏍囩偣鍓嶇殑瀛楋級鐪熺殑鎾畬浜嗐傝繖鏍锋墦鏂偣 + 钀藉湪鍙ユ湯鏍囩偣銆佽闊虫挱瀹岄偅涓鍒伙紝涓嶄細鎶婂彞瀛愪粠涓棿鎴柇銆 + """ + def __init__(self): + self._speaking = False + self._punct_seq = 0 # 鍙ユ湯鏍囩偣璁℃暟锛堟瘡鍑虹幇涓涓 +1锛 + + def update(self, text: str, audio_ms: float, is_listen: bool): + self._speaking = not is_listen + if text and any(p in text for p in _SENT_PUNCT): + self._punct_seq += 1 + + @property + def speaking(self) -> bool: + return self._speaking + + def snapshot(self): + return {"punct_seq": self._punct_seq} + + def boundary_reached(self, snap, speaker=None): + """鑷 snap 涔嬪悗锛氬嚭鐜颁簡鏂板彞鏈爣鐐癸紝涓旇娈佃闊冲凡鎾畬锛坰peaker 闃熷垪杩戠┖锛夈""" + if self._punct_seq <= snap["punct_seq"]: + return False # 杩樻病鍑虹幇鏂扮殑鍙ユ湯鏍囩偣 鈫 浠嶅湪淇濇姢褰撳墠鍙 + # 鍑虹幇浜嗘柊鏍囩偣锛氱瓑璇煶鐪熺殑鎾埌杩欓噷锛坰peaker 闃熷垪鍩烘湰娓呯┖锛 + if speaker is not None: + try: + if speaker.pending_ms() > 150.0: # 杩樻湁 >150ms 娌℃挱瀹 鈫 璇煶杩樻病鍒版爣鐐 + return False + except Exception: + pass + return True + +# 鈹鈹 寮哄埗鎺柦锛氭挱鏀剧绾块鐢熸垚鐨 reject 鎻愮ず wav锛堟ā鍨嬮煶鑹诧紝杩愯鏃朵笉纰 chat/KV锛夆攢鈹 +_REJECT_WAV_CACHE = {} # reason -> float32 ndarray锛堣繘绋嬪唴缂撳瓨锛屽彧璇讳竴娆$洏锛 + + +def _load_reject_wav(wav_dir, reason): + """璇婚鐢熸垚鐨 {reason}.wav锛24kHz 16-bit mono锛夆啋 float32 ndarray锛屽甫缂撳瓨銆 + 鎵句笉鍒拌繑鍥 None锛堣烦杩囨挱鎶ワ紝涓嶉樆鏂富娴佺▼锛夈""" + import wave as _wave + if reason in _REJECT_WAV_CACHE: + return _REJECT_WAV_CACHE[reason] + path = os.path.join(wav_dir, str(reason) + ".wav") + if not os.path.isfile(path): + _REJECT_WAV_CACHE[reason] = None + return None + try: + with _wave.open(path, "rb") as wf: + n = wf.getnframes() + raw = wf.readframes(n) + pcm16 = np.frombuffer(raw, dtype=np.int16) + pcm = (pcm16.astype(np.float32) / 32768.0) + _REJECT_WAV_CACHE[reason] = pcm + return pcm + except Exception: + _REJECT_WAV_CACHE[reason] = None + return None + +# 鈹鈹 寮哄埗鎺柦锛堥槻骞昏锛夛細reject 鏃剁敤妯″瀷闊宠壊蹇垫彁绀 鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹鈹 +# 澶嶇敤 test_tts_speak 楠岃瘉杩囩殑閾捐矾锛氳蛋 gateway 鐨 /ws/chat(wss)锛寊ero-shot TTS锛 +# 妯″瀷闊宠壊蹇典换鎰忔枃鏈紝杩斿洖 float32 24kHz 闊抽 鈫 PCSpeaker 鎾 +# 鍋/鎭㈠璧 8021(funnel.stop/funnel.resume)锛屽拰鐪熶汉銆屽仠涓涓嬨嶅悓涓鏉 controller銆 +_TTS_SYS_PROMPT = ( + "妯′豢闊抽鏍锋湰鐨勯煶鑹插苟鐢熸垚鏂扮殑鍐呭銆傝鐢ㄨ繖绉嶅0闊抽鏍兼潵涓虹敤鎴锋彁渚涘府鍔┿" + "鐩存帴浣滅瓟锛屼笉瑕佹湁鍐椾綑鍐呭銆" +) + + +async def _speak_hint_via_chat( + gateway_host: str, + gateway_port: int, + hint_text: str, + speaker, + ssl_ctx, +) -> bool: + """鐢 chat zero-shot TTS 璁╂ā鍨嬬敤鑷繁闊宠壊蹇 hint_text锛屾挱鍒 PCSpeaker銆 + + 杩斿洖 True=蹇靛嚭骞跺凡鍏ラ槦鎾斁锛汧alse=澶辫触锛堜笉闃绘柇涓绘祦绋嬶級銆 + """ + url = f"wss://{gateway_host}:{gateway_port}/ws/chat" + req = { + "messages": [ + {"role": "system", "content": _TTS_SYS_PROMPT}, + {"role": "user", "content": "璇锋湕璇讳互涓嬪唴瀹癸細" + hint_text}, + ], + "streaming": False, + "generation": {"max_new_tokens": 128, "length_penalty": 1.1}, + "tts": {"enabled": True}, + "use_tts_template": True, + "omni_mode": False, + } + try: + async with aiohttp.ClientSession() as sess: + async with sess.ws_connect(url, ssl=ssl_ctx, max_msg_size=128 * 1024 * 1024) as ws: + await ws.send_json(req) + audio_b64 = None + sr = 24000 + # 鍔犺秴鏃讹細蹇垫彁绀鸿蛋 /ws/chat,鑻 duplex 鍗犵潃 worker 浼氭帓闃熴 + # 瓒呮椂杩斿洖,閬垮厤姘镐箙闃诲 image_loop(鍚﹀垯鍚庣画鎶撳浘鍏ㄥ仠)銆 + deadline = time.monotonic() + 8.0 + while True: + remaining = deadline - time.monotonic() + if remaining <= 0: + LOG.warning("[寮哄埗鎺柦] 蹇垫彁绀鸿秴鏃(鍙兘 /ws/chat 鎺掗槦,duplex 鍗犵敤 worker)") + return False + try: + msg = await asyncio.wait_for(ws.receive(), timeout=remaining) + except asyncio.TimeoutError: + LOG.warning("[寮哄埗鎺柦] 蹇垫彁绀鸿秴鏃(绛 chat 鍝嶅簲)") + return False + if msg.type != aiohttp.WSMsgType.TEXT: + if msg.type in (aiohttp.WSMsgType.CLOSED, aiohttp.WSMsgType.ERROR): + return False + continue + data = json.loads(msg.data) + if data.get("type") == "done": + audio_b64 = data.get("audio_data") + sr = data.get("audio_sample_rate") or 24000 + break + if data.get("type") == "error": + LOG.warning("[寮哄埗鎺柦] chat TTS error: %s", data.get("error")) + return False + if not audio_b64: + LOG.warning("[寮哄埗鎺柦] chat TTS 鏃犻煶棰戣繑鍥") + return False + # audio_data 鏄 float32 base64锛24kHz mono锛夆啋 PCSpeaker(涔熸槸 24kHz float32) + import base64 as _b64 + pcm = np.frombuffer(_b64.b64decode(audio_b64), dtype=np.float32) + # block_and_flush 涔嬪悗 PCSpeaker.blocked=True锛宔nqueue 浼氳涓紱鍏 resume 瑙i櫎 + await speaker.resume() + await speaker.enqueue(pcm, generation=0) + LOG.info("[寮哄埗鎺柦] 蹇垫彁绀: %r (%d samples, %.2fs)", + hint_text, pcm.size, pcm.size / float(sr)) + return True + except Exception as e: + LOG.warning("[寮哄埗鎺柦] 蹇垫彁绀哄け璐: %s", e) + return False + + +# ============================================================ +# ESP32 闊抽杈撳叆锛歅C 涓诲姩杩 ESP32 鐨 /ws_audio_v2锛屾敹 int16 PCM +# 锛堟憳鑷 demo_esp32_duplex_0703 鐨 esp32_audio_reader锛屽幓鎺 ring/live_rec锛 +# 鐩存帴鎶婃瘡鍖呴煶棰戝杺缁 harness 楠ㄦ灦鐨 audio_queue + harness.send_audio锛 +# ============================================================ +async def esp32_audio_reader( + host: str, + port: int, + manager: GatewaySessionManager, + harness: HarnessClient, + audio_queue: DropOldestAudioQueue, + audio_mirror: AudioMirrorChunker, + stats: RuntimeStats, + input_gain: float, + stop_evt: asyncio.Event, + probe=None, + live_rec=None, +) -> None: + url = f"ws://{host}:{port}/ws_audio_v2" + LOG.info("[ESP32] audio WS: %s", url) + backoff = 1.0 + last_log = time.monotonic() + + while not stop_evt.is_set(): + try: + async with aiohttp.ClientSession() as session: + async with session.ws_connect(url, heartbeat=30, max_msg_size=0) as ws: + LOG.info("[ESP32] audio WS connected") + backoff = 1.0 + stats.audio_clients = 1 + async for msg in ws: + if stop_evt.is_set(): + break + if msg.type != aiohttp.WSMsgType.BINARY: + if msg.type in (aiohttp.WSMsgType.CLOSED, aiohttp.WSMsgType.ERROR): + break + continue + data = msg.data + if len(data) < ESP32_PKT_HDR: + continue + # 鍥轰欢鍖呭ご: seq(4) ts_ms(4) n_samples(2) reserved(2) + # 娉ㄦ剰: 绗4瀛楁鏄 reserved(pad)锛屼笉鏄涪鍖呮暟锛涚湡涓㈠寘鐢 seq 璺冲彉鎺ㄦ柇 + seq, ts_ms, n_samples, reserved = struct.unpack(" 0.0: + stats.non_silent_audio_packets += 1 + amplified = apply_pcm16_gain(raw, input_gain) + audio_queue.put_nowait(amplified) # -> gateway session + for frame in audio_mirror.feed(amplified): + await harness.send_audio(frame) # -> 8021 ASR 闀滃儚 + # -o record锛氬綍 user 闊筹紙鍘熷 samples锛屾湭鏀惧ぇ锛 + if live_rec is not None: + try: + live_rec.feed_user_raw(samples) + except Exception: + pass + + if probe is not None: + probe.mark_audio(seq, n_samples, packet_rms) + + now = time.monotonic() + if now - last_log >= 5.0: + LOG.info( + "[ESP32] rx=%d pkts seq=%d ts=%d rsv=%d rms=%.4f", + stats.audio_packets, seq, ts_ms, reserved, stats.audio_rms, + ) + last_log = now + except asyncio.CancelledError: + raise + except (aiohttp.ClientError, asyncio.TimeoutError, OSError) as e: + LOG.warning("[ESP32] audio WS error: %s, retry in %.1fs", e, backoff) + finally: + stats.audio_clients = 0 + if not stop_evt.is_set(): + try: + await asyncio.wait_for(stop_evt.wait(), timeout=backoff) + except asyncio.TimeoutError: + pass + backoff = min(backoff * 2, 10.0) + + LOG.info("[ESP32] audio reader stopped") + + +# ============================================================ +# rerun 闊抽 reader锛氳 live_user.wav锛屾寜 40ms 鍖呭杺 audio_queue + harness +# 锛堢Щ妞嶈嚜 rerun_source.local_pcm_reader锛屼笅娓镐粠 ring 鏀逛负 esp32 鐨 +# audio_queue + harness锛屽叾浣欒妭濂/鍥炴斁閫昏緫涓鑷达級 +# ============================================================ +async def rerun_audio_reader( + session_dir: str, + manager: GatewaySessionManager, + harness: HarnessClient, + audio_queue: DropOldestAudioQueue, + audio_mirror: AudioMirrorChunker, + stats: RuntimeStats, + input_gain: float, + stop_evt: asyncio.Event, + speed: float = 1.0, + live_rec=None, + ready_evt: Optional[asyncio.Event] = None, + replay_t0=None, + hard_stop: Optional[asyncio.Event] = None, + done_evt: Optional[asyncio.Event] = None, +) -> None: + if hard_stop is None: + hard_stop = stop_evt + # 绛 duplex session 灏辩华鍐嶆帹锛屽惁鍒欐帓闃熸湡闂寸殑闊抽鍏ㄤ涪锛堣 _wait_gateway_ready锛 + if ready_evt is not None: + await ready_evt.wait() + import wave as _wave + wav_path = os.path.join(session_dir, "live_user.wav") + if not os.path.isfile(wav_path): + LOG.error("[RERUN] live_user.wav 涓嶅瓨鍦: %s", wav_path) + stop_evt.set() + return + with _wave.open(wav_path, "rb") as w: + sr_in = w.getframerate() + pcm_i16 = np.frombuffer(w.readframes(w.getnframes()), dtype=np.int16) + _SR = 16000 + _PKT = 640 # 40ms @16k + total = pcm_i16.size + LOG.info("[RERUN] audio: %s samples=%d dur=%.2fs sr=%d", + wav_path, total, total / _SR, sr_in) + + packet_interval = (_PKT / _SR) / max(speed, 0.01) + ts_ms = 0 + cursor = 0 + pkt = 0 + # next_tick 蹇呴』鏄"鏈嚱鏁扮湡姝e紑濮嬫帹鍖呯殑閭d竴鍒"銆 + # 鏇剧粡杩欓噷鍐欐垚 replay_t0()锛堝氨缁棬鏀捐鐨勬椂鍒伙級锛屼絾閭d箣鍚庤繕瑕佺粡杩囦换鍔¤皟搴︺ + # 璇 wav銆佹墦鏃ュ織锛岀瓑璺戝埌寰幆閲屾椂 next_tick 宸茬粡鏄繃鍘绘椂鍒 鈫 + # sleep_for 鎭掍负璐 鈫 姣忚疆閮戒笉 sleep 鈫 564 涓寘鍦ㄥ嚑鍗佹绉掑唴鍏ㄧ亴杩 audio_queue + # 鈫 闃熷垪瀹归噺 96銆丏ropOldest 鎶婂紑澶寸殑鎻愰棶鍏ㄤ涪浜嗭紝妯″瀷鍙惉鍒版渶鍚 3.8 绉掔殑灏鹃煶銆 + # 锛坰tats 閲岀殑 audio=25.0pps 缁熻鐨勬槸"鎺ㄨ繘闃熷垪"鐨勬暟閲忥紝姝eソ鎺╃洊浜嗚繖涓棶棰樸傦級 + # 鍥鹃偅杈规寜 frames.jsonl 鐨 t 绛夊緟銆佷互 replay_t0 涓洪浂鐐癸紝涓よ竟璧风偣宸嚑鍗佹绉掞紝 + # 瀵 1s 绮掑害鐨 chunk 娌℃湁褰卞搷銆 + next_tick = time.monotonic() + # 鍙彈 hard_stop 鎺у埗锛屼笉鍐嶈 stop_evt 鎺愭柇銆 + # 鍘熸潵鏄 `while cursor < total and not stop_evt.is_set()`锛 + # 鍥鹃偅鏉¤矾寰勬斁瀹屽浘浼 set stop_evt锛岃岀礌鏉愮殑鍥捐疆鏁板父灏戜簬闊抽绉掓暟 + # 锛坉rugbox 16 杞浘 vs 22.5s 闊抽锛夆啋 鍥惧厛鏀惧畬灏辨妸闊抽鎺ㄩ佹帎浜嗭紝 + # 鍚庡崐娈垫彁闂牴鏈病杩涙ā鍨嬨備袱鏉℃祦搴斿悇鑷窇瀹岃嚜宸辩殑銆 + while cursor < total and not hard_stop.is_set(): + end = min(cursor + _PKT, total) + chunk_i16 = pcm_i16[cursor:end] + cursor = end + pkt += 1 + raw = chunk_i16.astype(" 0: + # 杩欓噷绛夌殑鏄 hard_stop锛屼笉鏄 stop_evt銆 + # 涔嬪墠 while 鏉′欢鏀规垚浜 hard_stop锛屼絾寰幆浣撻噷浠 await stop_evt 骞 break锛 + # 绛変簬娌℃敼鈥斺斿浘鏀惧畬 set stop_evt 鐓ф牱鎶婇煶棰戞帹閫佹帎鏂 + try: + await asyncio.wait_for(hard_stop.wait(), timeout=sleep_for) + break + except asyncio.TimeoutError: + pass + + if done_evt is not None: + done_evt.set() + LOG.info("[RERUN] audio 鎺ㄥ畬 %d 鍖 (%.2fs)锛岀瓑妯″瀷璇村畬鈥", pkt, ts_ms / 1000.0) + # 闊抽鎺ㄥ畬涓嶇珛鍗 stop銆傛敞鎰忥細杩欓噷**涓嶈兘** await stop_evt 灏 break 鈥斺 + # 鍥鹃偅鏉¤矾寰勬斁瀹屽浘涔熶細 set stop_evt锛屼袱鏉′簰鐩歌俯锛岀粨鏋滆皝鍏堝埌璋佹妸鍙︿竴鏉℃帎浜 + # 锛堝浘灏戠殑绱犳潗灏ゅ叾鏄庢樉锛氬浘寰堝揩鏀惧畬 鈫 鐩存帴鏀跺伐 鈫 妯″瀷璇濇病璇村畬锛夈 + # 杩欓噷鐙珛鎸夋椂闂寸瓑锛岃妯″瀷鏈夋満浼氭妸鏈鍚庝竴鍙ヨ瀹岋紱鐪熸鐨勭粨鏉熺敱 image_loop + # 鐨勯潤榛樺垽鎹 + 灏惧反鍐冲畾锛屾垨璧拌繖閲岀殑鍏滃簳銆 + waited = 0.0 + while waited < 120.0: + if stop_evt.is_set(): + # 鍙︿竴鏉″凡鍒ゅ畾缁撴潫锛屽啀瀹介檺涓灏忔璁 TTS 鎾畬 + await asyncio.sleep(2.0) + break + await asyncio.sleep(0.5) + waited += 0.5 + if not stop_evt.is_set(): + stop_evt.set() + LOG.info("[RERUN] audio reader stopped") + + +# ============================================================ +# rerun 鍥惧儚 loop锛氱敤 LocalImageSource 鎸 chunk 椤哄簭璇 best 鍥撅紝鐩存帴鍠傛ā鍨 +# 锛-o rerun 涓嶉噸璺 funnel鈥斺攂est 宸叉槸绛涜繃鐨勶紝鐩存帴 send_frame锛 +# ============================================================ +async def rerun_image_loop( + session_dir: str, + latest_frame, + harness: HarnessClient, + stats: RuntimeStats, + interval_s: float, + stop_evt: asyncio.Event, + web_ui=None, + live_rec=None, + ready_evt: Optional[asyncio.Event] = None, + replay_t0=None, + done_evt: Optional[asyncio.Event] = None, + peer_done: Optional[asyncio.Event] = None, + speaker=None, + sent_tracker=None, + tail_wait_s: float = 30.0, +) -> None: + if ready_evt is not None: + await ready_evt.wait() + try: + from .rerun_source import LocalImageSource + except Exception as e: + LOG.error("[RERUN] 鏃犳硶瀵煎叆 LocalImageSource: %s", e) + stop_evt.set() + return + from pathlib import Path as _Path + try: + src = LocalImageSource(_Path(session_dir)) + except Exception as e: + LOG.error("[RERUN] LocalImageSource 鍒濆鍖栧け璐: %s", e) + stop_evt.set() + return + # 鎸 chunk_idx 鍗囧簭鍥炴斁 + idxs = sorted(src._map.keys()) + LOG.info("[RERUN] image loop: %d 甯у緟鍥炴斁", len(idxs)) + for cidx in idxs: + if stop_evt.is_set(): + break + jpeg = await src.capture(cidx) + if jpeg: + ts = now_ms() + latest_frame.set(jpeg, ts) + stats.image_count += 1 + await harness.send_frame(jpeg, latest_frame.sequence, ts) + LOG.info("[RERUN] send frame idx=%d (%d bytes)", cidx, len(jpeg)) + _emit_chunk_img(web_ui, jpeg, latest_frame.sequence, True) + if live_rec is not None: + try: + live_rec.on_frame(jpeg, latest_frame.sequence) + except Exception: + pass + try: + await asyncio.wait_for(stop_evt.wait(), timeout=interval_s) + break + except asyncio.TimeoutError: + pass + LOG.info("[RERUN] image loop 鍥炴斁瀹屾瘯") + if done_evt is not None: + done_evt.set() + # -o rerun 鐨勬敹灏俱備箣鍓嶈繖鏉¤矾寰勫浘鏀惧畬灏变粈涔堥兘涓嶅仛锛岀粨鏉熷叏闈 + # rerun_audio_reader 閲 `while waited < 120.0` 鐨勫厹搴曞共绛変袱鍒嗛挓 + #锛堟棩蹇楄〃鐜帮細闊抽鎺ㄥ畬鍚庝竴涓 audio=0.0pps 鐨 STATS锛屼袱鍒嗛挓鎵嶉锛夛紝 + # 鑰屼笖 --rerun-tail-wait-s 瀵瑰畠鏃犳晥銆 + # 鍒ゅ畾涓 esp32_image_loop 淇濇寔涓鑷达細鍏堢瓑闊抽涔熸帹瀹岋紝鍐嶇湅妯″瀷璇村畬娌℃湁 + #锛堥槦鍒楃┖ 涓 涓嶅湪 speak锛岃繛缁繚鎸 2s 鎵嶇畻锛夛紝鏈鍚庡姞 2s 灏惧反銆 + if peer_done is not None and not peer_done.is_set(): + LOG.info("[RERUN] 鍥炬斁瀹岋紝绛夐煶棰戞帹瀹屸") + try: + await asyncio.wait_for(peer_done.wait(), timeout=180.0) + except asyncio.TimeoutError: + LOG.warning("[RERUN] 绛夐煶棰戣秴鏃") + _QUIET_HOLD_S, _TAIL_S = 2.0, 2.0 + _quiet_since = None + _deadline = time.monotonic() + tail_wait_s + while time.monotonic() < _deadline and not stop_evt.is_set(): + try: + pending = speaker.pending_ms() if speaker is not None else 0.0 + except Exception: + pending = 0.0 + speaking = bool(getattr(sent_tracker, "speaking", False)) if sent_tracker else False + if pending <= 50 and not speaking: + if _quiet_since is None: + _quiet_since = time.monotonic() + elif time.monotonic() - _quiet_since >= _QUIET_HOLD_S: + break + else: + _quiet_since = None + await asyncio.sleep(0.25) + try: + await asyncio.wait_for(stop_evt.wait(), timeout=_TAIL_S) + except asyncio.TimeoutError: + pass + LOG.info("[RERUN] 缁撴潫锛堥潤榛樹繚鎸%.1fs + 灏惧反%.1fs锛", _QUIET_HOLD_S, _TAIL_S) + stop_evt.set() + + +# ============================================================ +# ESP32 鍥惧儚杈撳叆锛歍CP 5000 鎸佷箙杩炴帴锛岃姹-鍝嶅簲鍙栬8 JPEG +# 锛堟憳鑷 demo_esp32_duplex_0703 鐨 TcpImageClient锛屾帴鍙d笉鍙橈級 +# 甯уご(20B 灏忕): magic(4) frame_id(4) w(2) h(2) fmt(1) reserved(3) len(4) +# ============================================================ +_ROTATE_MAP = None + + +def _rotate_jpeg(jpeg: bytes, deg: int) -> bytes: + """鎶婄浉鏈哄師濮嬪抚鎸夐『鏃堕拡 deg 杞銆俤eg 鈭 {0,90,180,270}锛0 鐩存帴鍘熸牱杩斿洖銆 + + 涓轰粈涔堣鍦ㄨ繖閲岃浆锛氱溂闀滅殑鎽勫儚澶存槸**鐗╃悊渚ц/鍊掕**鐨勶紙devices.json 鐨 rotate锛夛紝 + 鍑烘潵鐨勫抚鏈韩灏辨槸姝殑銆備笉杞鐨勮瘽锛屾紡鏂楃殑鏂瑰悜鍒嗙被鍣ㄤ細鎶"鐩告満渚ц" + 璇垽鎴"鐢ㄦ埛鎶婄洅瀛愭嬁鍙嶄簡"骞舵彁绀虹敤鎴风炕杞 鈥斺 鐢ㄦ埛缈讳簡鍙嶈屾洿姝 + 鍙栨櫙鎴柇鐨勪笂涓嬪乏鍙宠涔夈乢PAN_HINT 鐨"鍚戜笂鐪/鍚戜笅鐪"涔熷叏閮戒細鍙嶃 + 鎵浠ュ繀椤诲湪**杩涙紡鏂椾箣鍓**杞紝杞硶鐓ф惉 demo_esp32_duplex_0703銆 + + 娉ㄦ剰 PIL 鎸夐嗘椂閽堢畻锛氶『鏃堕拡 90掳 瑕佺敤 ROTATE_270銆 + """ + if not deg or not jpeg: + return jpeg + global _ROTATE_MAP + try: + from PIL import Image + import io + except Exception: + return jpeg + if _ROTATE_MAP is None: + _ROTATE_MAP = { + 90: Image.Transpose.ROTATE_270, # 椤烘椂閽 90掳 = PIL 鐨 ROTATE_270 + 180: Image.Transpose.ROTATE_180, + 270: Image.Transpose.ROTATE_90, + } + op = _ROTATE_MAP.get(int(deg) % 360) + if op is None: + return jpeg + try: + im = Image.open(io.BytesIO(jpeg)) + im = im.transpose(op) + buf = io.BytesIO() + im.save(buf, format="JPEG", quality=92) + return buf.getvalue() + except Exception as e: + LOG.warning("[ROTATE] 杞澶辫触(%d掳)锛岀敤鍘熷浘: %r", deg, e) + return jpeg + + +class TcpImageClient: + def __init__(self, host: str, port: int = 5000, rotate: int = 0): + self.host = host + self.port = port + self.rotate = int(rotate) % 360 + self._reader: Optional[asyncio.StreamReader] = None + self._writer: Optional[asyncio.StreamWriter] = None + self._lock = asyncio.Lock() + + async def _ensure_conn(self, timeout_s: float) -> bool: + if self._reader is not None and self._writer is not None and not self._writer.is_closing(): + return True + try: + self._reader, self._writer = await asyncio.wait_for( + asyncio.open_connection(self.host, self.port), timeout=timeout_s) + sock = self._writer.get_extra_info("socket") + if sock is not None: + import socket as _s + sock.setsockopt(_s.IPPROTO_TCP, _s.TCP_NODELAY, 1) + LOG.info("[TCP-IMG] connected to %s:%d", self.host, self.port) + return True + except Exception as e: + LOG.debug("[TCP-IMG] connect failed: %s", e) + await self._close() + return False + + async def _close(self) -> None: + if self._writer is not None: + try: + self._writer.close() + except Exception: + pass + self._reader = None + self._writer = None + + async def capture(self, timeout_s: float = 1.0) -> Optional[bytes]: + async with self._lock: + if not await self._ensure_conn(timeout_s): + return None + try: + self._writer.write(b"C") + await self._writer.drain() + hdr = await asyncio.wait_for(self._reader.readexactly(20), timeout=timeout_s) + magic, frame_id, w, h = struct.unpack_from(" 4 * 1024 * 1024: + LOG.warning("[TCP-IMG] insane len=%d, reconnect", length) + await self._close() + return None + data = await asyncio.wait_for(self._reader.readexactly(length), timeout=timeout_s) + # 杞鍚庡啀浜ょ粰涓婂眰锛歭atest_frame / 婕忔枟 / 褰曞埗鎷垮埌鐨勯兘鏄鍥撅紝 + # rerun 鏃朵笉蹇呭啀杞紙鍚﹀垯浼氳浆涓ゆ锛夈 + return _rotate_jpeg(bytes(data), self.rotate) + except asyncio.CancelledError: + raise + except Exception as e: + LOG.warning("[TCP-IMG] capture failed: %r, reconnect", e) + await self._close() + return None + + +# ============================================================ +# funnel rerun锛氭妸褰曞埗鐨勬暣绨囧甯у綋鍥炬簮锛屽鐢 esp32_image_loop 閲嶈窇婕忔枟+鎾姤 +# RecordedImageClient.capture() 鎸 frames.jsonl 姣忚疆鏁寸皣椤哄簭鍚愬抚锛 +# funnel.run_once 璋 N 娆″噾涓绨 鈫 閲嶅垽 send/reject 鈫 鎾姤/鏍囩偣淇濇姢锛岄昏緫鍏ㄥ鐢ㄣ +# ============================================================ +class RecordedImageClient: + def __init__(self, session_dir: str, n_frames: int, no_funnel: bool = False): + import json as _json + self.dir = session_dir + self.images_dir = os.path.join(session_dir, "images") + self.n_frames = max(1, int(n_frames)) + self._rounds = [] # 姣忚疆: [jpg_bytes, ...] + fr_path = os.path.join(session_dir, "frames.jsonl") + self._round_names = [] # 姣忚疆鐨勫抚鏂囦欢鍚嶏紙渚 record 澶嶇敤锛 + self._round_t = [] # 姣忚疆褰曞埗鏃跺埢 t锛堢锛夛紝鐢ㄤ簬鎸夊師濮嬫椂闂磋酱鍥炴斁 + with open(fr_path, encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + rec = _json.loads(line) + except Exception: + continue + names = rec.get("frames") or [] + jpgs = [] + keep_names = [] + for nm in names: + p = os.path.join(self.images_dir, nm) + if os.path.isfile(p): + with open(p, "rb") as fp: + jpgs.append(fp.read()) + keep_names.append(nm) + if jpgs: + self._rounds.append(jpgs) + self._round_names.append(keep_names) + self._round_t.append(float(rec.get("t", len(self._rounds) - 1))) + LOG.info("[RERUN-IMG] 杞藉叆 %d 杞紝姣忚疆鈮%d甯%s", + len(self._rounds), self.n_frames, + "锛堟棤婕忔枟锛氭瘡杞彇涓棿寮犱唬琛ㄥ抚锛" if no_funnel else "锛坒unnel锛氭暣绨囬愬抚锛") + self._round_i = 0 + self._frame_i = 0 + self._no_funnel = no_funnel + self.exhausted = False + self._t0 = None # 鍥炴斁璧风偣锛堜紭鍏堢敤澶栭儴娉ㄥ叆鐨勭粺涓闆剁偣锛 + self.replay_t0 = None # 鐢 runtime 娉ㄥ叆锛屼笌闊抽鍏辩敤 + # 鏁翠唤 record 涓寮犲浘閮芥病鏈夛紙鍥惧儚閾捐矾褰撴椂涓簡 / 鎵嬪姩鍒犵┖ images锛夛細 + # 杩欐槸鏈夋晥绱犳潗锛堝鐓э細鏃犳紡鏂椻啋妯″瀷 0 鍥剧紪鍦烘櫙锛涙湁婕忔枟鈫掑叏绋 reject no_frames锛夈 + # 姝ゆ椂涓嶈兘缃 exhausted锛屽惁鍒欎細琚綋鎴"鍥炬斁瀹屼簡"绉掗锛涜闊抽鎺ㄥ畬鏉ョ粨鏉熸暣鍦恒 + self._no_images = (len(self._rounds) == 0) + if self._no_images: + LOG.info("[RERUN-IMG] 杩欎唤 record 娌℃湁鍙敤鍥惧儚 鈫 鍏ㄧ▼鏃犲浘鍥炴斁" + "锛堢敱闊抽闀垮害鍐冲畾缁撴潫锛") + + def current_round_cluster(self): + """杩斿洖褰撳墠杞殑鏁寸皣 (jpg_list, name_list)锛屼緵鏃犳紡鏂 rerun 瀛 record 鐢ㄣ""" + i = self._round_i + if 0 <= i < len(self._rounds): + return self._rounds[i], self._round_names[i] + return [], [] + + async def _wait_round_time(self): + """鎸夊綍鍒舵椂鍒诲洖鏀撅細绛夊埌澧欓挓璧板埌鏈疆鐨 t 鍐嶅悙杩欎竴杞殑甯с + + 褰曞埗鏃舵瘡杞楁椂骞朵笉鍧囧寑锛堝鐒﹁疆 ~1.9s銆乺eject 鎾姤杞 2.3~4s锛夛紝鑻 rerun 鎸 + image_loop 鐨勫浐瀹 interval_s 鍖閫熷悆鍥撅紝灏变細浠ユ暟鍊嶉熸妸鍥惧悆鍏夛紙瀹炴祴 33s 鐨勫綍鍒 + 11s 灏辨斁瀹岋級銆傞煶棰戞槸鎸夌湡瀹炴椂闀垮洖鏀剧殑锛岀敤 frames.jsonl 鐨 t 瀵归綈锛屼袱杈瑰悓涓鏉 + 鏃堕棿杞淬 + """ + if self._t0 is None: + # 浼樺厛鐢ㄧ粺涓鍥炴斁闆剁偣锛堜笌闊抽鍚屾簮锛夛紱娌℃湁鎵嶉鍥"绗竴娆 capture 閭d竴鍒" + self._t0 = self.replay_t0 if self.replay_t0 is not None else time.monotonic() + i = self._round_i + if 0 <= i < len(self._round_t): + target = self._round_t[i] + delay = target - (time.monotonic() - self._t0) + if delay > 0: + await asyncio.sleep(min(delay, 10.0)) + + async def capture(self, timeout_s: float = 1.0): + if self._no_images: + # 鍏ㄧ▼鏃犲浘锛氫竴鐩磋繑鍥 None锛屼絾涓嶇疆 exhausted锛堜笉鎻愬墠缁撴潫锛夈 + # 鏈夋紡鏂 鈫 run_once 鎷垮埌 0 甯 鈫 reject(no_frames) 鈫 鎾"娌℃湁鎷垮埌鐢婚潰"锛 + # 鏃犳紡鏂 鈫 涓 send 鈫 妯″瀷 0 鍥 鈫 鏆撮湶缂栧満鏅 + await asyncio.sleep(0.05) + return None + if self._round_i >= len(self._rounds): + self.exhausted = True + return None + # 鍙湪涓杞殑绗竴甯т笂绛夊緟锛屽悓涓杞唴鐨勫甯ц繛缁悙锛堟ā鎷熷師濮嬮珮棰戞姄甯э級 + if self._frame_i == 0: + await self._wait_round_time() + rnd = self._rounds[self._round_i] + if self._no_funnel: + # 鏃犳紡鏂楋紙瀵圭収缁勶級锛氭瘡杞彇涓寮犱唬琛ㄥ抚锛堜腑闂村紶锛屾ā鎷"闅忔墜鎶撲竴寮犲氨鍙"锛夛紝 + # 姣忔 capture 鐩存帴杩涘叆涓嬩竴杞傚惈妯$硦鍥句細鐓у彂 鈫 鏆撮湶骞昏銆 + jpg = rnd[len(rnd) // 2] + self._round_i += 1 + return jpg + # funnel锛氫竴甯у抚鍚愭暣绨囷紝渚 run_once 璋 N 娆″噾涓绨 + jpg = rnd[self._frame_i % len(rnd)] + self._frame_i += 1 + if self._frame_i >= self.n_frames: + self._frame_i = 0 + self._round_i += 1 + return jpg + + async def _close(self): + pass + + +# ============================================================ +# ESP32 鍥惧儚杞寰幆锛氬畾鏃剁敤 TCP 鍙栧浘 -> latest_frame + harness.send_frame +# ============================================================ +async def esp32_image_loop( + client: TcpImageClient, + latest_frame: LatestFrame, + harness: HarnessClient, + stats: RuntimeStats, + interval_s: float, + timeout_s: float, + stop_evt: asyncio.Event, + probe=None, + funnel=None, + recorder=None, + speaker=None, + gateway_host: str = "127.0.0.1", + gateway_port: int = 8040, + ssl_ctx=None, + force_measure: bool = False, + manager=None, + reject_wav_dir: str = "assets/reject_wav", + sent_tracker=None, + live_rec=None, + web_ui=None, + ready_evt: Optional[asyncio.Event] = None, + tail_wait_s: float = 30.0, + no_reject: bool = False, + peer_done: Optional[asyncio.Event] = None, + done_evt: Optional[asyncio.Event] = None, +) -> None: + if ready_evt is not None: + await ready_evt.wait() + LOG.info("[ESP32] image loop start (interval=%.2fs, funnel=%s)", + interval_s, "on" if funnel else "off") + _last_grab_mono = None + + # 鈹鈹 寮哄埗鎺柦鐘舵 鈹鈹 + _funnel_stopped = False # 鏄惁宸插彂杩 funnel.stop锛堝湪鍋滀綇鎬侊紝绛夊ソ鍥 resume锛 + _hint_playing = False # 鎾姤杩涜涓紙鍚庡彴浠诲姟锛夛紝鏈熼棿涓嶉噸澶嶈Е鍙戙佷笉閫佸浘 + _last_hint_mono = 0.0 # 涓婃蹇垫彁绀虹殑鏃跺埢锛堣妭娴佺敤锛 + _HINT_THROTTLE_S = 5.0 # 鍚岀被杩炵画 reject 鐨勫康鎻愮ず鑺傛祦闂撮殧 + _pending_reject = None # 鏍囩偣淇濇姢锛氭寕璧蜂腑鐨 reject锛堢瓑鏍囩偣/瓒呮椂鍐嶅鏌ワ級 + # 鈹鈹 銆屾寔缁潖銆嶆娴嬶細3s 绐楀彛鍐 reject鈮2 鈫 鍒ゅ畾杈撳嚭鍩轰簬鍧忓浘=骞昏锛岀洿鎺ユ墦鏂+鍘嬪埗 鈹鈹 + # 瑕嗙洊"閿欓敊閿欏"鍦烘櫙锛堝墠鍗婁篃閿欙紝鏍囩偣淇濇姢涓嶉傜敤锛夈傜獥鍙e崟浣=鍒ゅ畾/chunk(姣忊増1s)銆 + # 渚濇嵁锛氭ā鍨嬫嬁鍥锯啋杈撳嚭绾 1-1.5s锛2s 閮藉潖鍒欏够瑙夊繀宸插湪杈撳嚭锛屽繀椤婚棴鍢淬 + _judge_history = [] # 鏈杩戝垽瀹氭椂鍒+鏄惁reject: [(mono, is_reject), ...] + _WINDOW_S = 3.0 # 绐楀彛闀垮害 + _BAD_THRESH = 2 # 绐楀彛鍐 reject鈮ユ 鈫 鎸佺画鍧忥紝鐩存帴鎵撴柇(璺宠繃鏍囩偣淇濇姢) + # _funnel_active锛氬康鎻愮ず涓寸晫鍖烘爣蹇楋紙棰勭暀缁欏厹搴曪細蹇垫彁绀洪偅鍑犵鐪熶汉鍠婅瘽鐨勫鐞嗭級 + # TODO(鍏滃簳,涓嬫瀹炵幇): 蹇垫彁绀烘湡闂磋嫢鏀跺埌鐪熶汉 STOP/RESUME/RESET锛 + # 鍏堟帎鎺夋彁绀洪煶(speaker.block_and_flush)+娓呮鏍囧織锛屽啀鎵ц鐪熶汉鎸囦护锛岄伩鍏嶆挒杞︺ + # 褰撳墠鍙崰浣嶏紝涓嶅疄鐜伴昏緫銆 + _funnel_active = False # noqa: F841 (棰勭暀) + + async def _capture_one(): + return await client.capture(timeout_s=timeout_s) + + while not stop_evt.is_set(): + t0 = time.monotonic() + # funnel rerun锛氬綍鍒剁殑鏁寸皣澶氬抚鏀惧畬浜 鈫 浼橀泤缁撴潫锛堜笉鎶 no_frames锛 + if getattr(client, "exhausted", False): + LOG.info("[FUNNEL-RERUN] 褰曞埗鍥惧凡鏀惧畬") + # 鍥捐疆鏁板父灏戜簬闊抽绉掓暟锛坉rugbox 16 杞 vs 22.5s锛夛紝涓嶈兘鍥句竴鏀惧畬灏辨敹宸ワ紝 + # 鍚﹀垯鍚庡崐娈甸煶棰戯紙鍙兘姝e惈鎻愰棶锛夋牴鏈病鎺ㄧ粰妯″瀷銆傚厛绛夐煶棰戜篃鎺ㄥ畬銆 + if peer_done is not None and not peer_done.is_set(): + LOG.info("[FUNNEL-RERUN] 绛夐煶棰戞帹瀹屸") + try: + await asyncio.wait_for(peer_done.wait(), timeout=180.0) + except asyncio.TimeoutError: + LOG.warning("[FUNNEL-RERUN] 绛夐煶棰戣秴鏃") + LOG.info("[FUNNEL-RERUN] 涓ゆ潯娴侀兘瀹屼簨锛岀瓑妯″瀷璇村畬鍚庣粨鏉") + # 绛夋ā鍨嬫妸璇濊瀹屽啀 stop銆傚垽鎹笉鑳藉彧鐪"闃熷垪绌"鈥斺旀ā鍨嬫槸娴佸紡浜у嚭鐨勶紝 + # 鏌愪竴鐬棿闃熷垪鎺掔┖鍙槸"涓嬩竴娈佃繕娌″埌"锛屼笉浠h〃杩欒疆缁撴潫锛堝浘灏戠殑绱犳潗灏ゅ叾鏄庢樉锛 + # 鍥惧緢蹇斁瀹岋紝鎭板ソ鎾炰笂闃熷垪绌猴紝灏辨妸杩樻病璇村畬鐨勮瘽鎺愪簡锛夈 + # 鏀逛负锛氶槦鍒楀繀椤昏繛缁 _QUIET_HOLD_S 淇濇寔绌猴紝涓 sent_tracker 涓嶅湪 speak 涓紝 + # 鍐嶅姞涓娈靛浐瀹氬熬宸达紝缁欐渶鍚庝竴鍙 TTS 鐣欏嚭鎾斁鏃堕棿銆 + _QUIET_HOLD_S = 2.0 + _TAIL_S = 2.0 + _quiet_since = None + _deadline = time.monotonic() + tail_wait_s + while time.monotonic() < _deadline and not stop_evt.is_set(): + try: + pending = speaker.pending_ms() if speaker is not None else 0.0 + except Exception: + pending = 0.0 + speaking = False + if sent_tracker is not None: + speaking = bool(getattr(sent_tracker, "speaking", False)) + if pending <= 50 and not speaking: + if _quiet_since is None: + _quiet_since = time.monotonic() + elif time.monotonic() - _quiet_since >= _QUIET_HOLD_S: + break + else: + _quiet_since = None + await asyncio.sleep(0.25) + # 灏惧反锛氳鏈鍚庝竴鍙 TTS 鎾畬锛屼篃缁欐ā鍨嬩竴鐐硅ˉ鍏呯殑鏈轰細 + try: + await asyncio.wait_for(stop_evt.wait(), timeout=_TAIL_S) + except asyncio.TimeoutError: + pass + LOG.info("[FUNNEL-RERUN] 缁撴潫锛堥潤榛樹繚鎸%.1fs + 灏惧反%.1fs锛", + _QUIET_HOLD_S, _TAIL_S) + stop_evt.set() + break + try: + if funnel is None: + # 鈹鈹 鏃犳紡鏂楋細鍙 1 甯х洿鍙戯紙鍚ā绯婂浘锛屾毚闇插够瑙夛級鈹鈹 + # record 浠嶅瓨鏁寸皣锛坢p4 鐢ㄥ叏閮ㄩ噰闆嗗抚锛屽拰鏈夋紡鏂椾竴鑷达紝鍙槸鏃 reject 鏍囨敞锛 + cluster_jpgs, cluster_names = ([], []) + if hasattr(client, "current_round_cluster"): + cluster_jpgs, cluster_names = client.current_round_cluster() + jpeg = await client.capture(timeout_s=timeout_s) + grab_ms = (time.monotonic() - t0) * 1000.0 + if jpeg: + ts = now_ms() + latest_frame.set(jpeg, ts) + stats.image_count += 1 + await harness.send_frame(jpeg, latest_frame.sequence, ts) + _emit_chunk_img(web_ui, jpeg, latest_frame.sequence, True) + if probe is not None: + since_last = (t0 - _last_grab_mono) * 1000.0 if _last_grab_mono else 0.0 + probe.mark_grab(latest_frame.sequence, grab_ms, len(jpeg), since_last) + _last_grab_mono = t0 + # 瀛 record锛堟棤婕忔枟锛歴end=True 鏃 reason锛夈 + # rerun 鏃跺浘婧愭槸 RecordedImageClient锛屾湁 current_round_cluster()锛 + # 鑳芥嬁鍒版暣绨囷紱**瀹炴満鏃跺浘婧愭槸 TcpImageClient锛屾病鏈夎繖涓柟娉**锛 + # cluster_jpgs 鎭掍负绌 鈥斺 鍘熸潵鍐 `and cluster_jpgs` 灏辨妸瀹炴満鐨 + # 褰曞埗鏁翠釜璺宠繃浜嗭紙瀹炴祴 frames.jsonl 0 rounds銆乮mages/ 绌猴級銆 + # 瀹炴満 no_funnel 鏈潵灏辨槸姣忚疆涓寮犲浘锛屾嬁涓嶅埌鏁寸皣灏辩敤杩欎竴甯у綋涓杞 + if live_rec is not None: + _frames = cluster_jpgs if cluster_jpgs else [jpeg] + _dec = type("D", (), {"frames": _frames, + "best_index": len(_frames) // 2, + "send": True, "reason": ""})() + try: + await asyncio.to_thread( + live_rec.on_funnel_round, _dec, latest_frame.sequence) + # best 鍥句篃瀛樹竴浠斤紙events.jsonl + img_*.jpg锛屼緵 -o rerun锛 + await asyncio.to_thread( + live_rec.on_frame, jpeg, latest_frame.sequence) + except Exception as e: + LOG.warning("[LIVE] on_funnel_round(no-funnel) err: %s", e) + else: + # 鈹鈹 鏈夋紡鏂楋細涓杞垽瀹氾紝鍚堟牸鎵 send_frame 鈹鈹 + decision = await funnel.run_once(_capture_one) + # 缁熶竴 record锛氭瘡杞暣绨囧甯 + 鍒ゅ畾 鈫 live_rec锛堜緵涓夌 rerun锛夈 + # 蹇呴』鏀惧湪鎵鏈夊垎鏀箣鍓嶏細鎸佺画鍧忓垎鏀湯灏炬湁 continue锛屾斁鍦ㄦ渶鍚庝細鎶 + # 瑙﹀彂鍘嬪埗鐨勯偅浜涜疆鏁寸皣婕忓綍锛堝疄娴 42s 鍙惤涓 8 杞級銆傚綍鍒舵槸鍘熷绱犳潗锛 + # 涓嶈鍥犱负璧颁簡鍝潯澶勭悊璺緞鑰岀己澶便 + if live_rec is not None: + try: + await asyncio.to_thread( + live_rec.on_funnel_round, decision, latest_frame.sequence) + except Exception as e: + LOG.warning("[LIVE] on_funnel_round err: %s", e) + grab_ms = decision.timings.get("grab_ms", 0.0) + async def _do_reject_interrupt(reason): + """鐪熸鎵ц鎵撴柇锛氬仠 duplex 鈫 鎾 wav 鈫 鎭㈠銆 + + **鍦ㄥ悗鍙颁换鍔¢噷璺戯紝涓嶈兘璁 image_loop 鍚屾绛夊畠**锛 + 閲岄潰鏈 sleep(0.35) + sleep(wav鏃堕暱+0.2)锛屽悎璁 3~4 绉掋 + 鍘熸潵鏄湪寰幆浣撻噷 await 鐨勶紝杩 3~4 绉掑唴涓嶆姄鍥俱佷笉鍒ゅ畾銆 + **涔熶笉鎺ㄧ涓瑙嗚** 鈥斺 鐢婚潰璧板嚑甯у氨瀹氫綇锛屾鏄 鈶 鐙湁鐨勭幇璞° + 鏇寸碂鐨勬槸瀹冧細鑷攣锛氭挱鎶ユ媺闀夸簡甯ч棿闅 鈫 鍏夋祦鎸 4.7 绉掔殑浣嶇Щ绠 鈫 + 鍚屾牱鐨勯潤姝㈢敾闈㈠垽鎴 severe_shake 鈫 鍙堟挱鎶 鈫 闂撮殧鍙堝彉闀裤 + 鏀规垚鍚庡彴璺戜箣鍚庯紝鍙栧浘鍜屾帹鍥剧収甯革紝鎾姤骞惰杩涜銆 + """ + nonlocal _hint_playing + _hint_playing = True + try: + await _do_reject_interrupt_inner(reason) + finally: + _hint_playing = False + + async def _do_reject_interrupt_inner(reason): + wav_pcm = _load_reject_wav(reject_wav_dir, reason) + if wav_pcm is None: + LOG.warning("[寮哄埗鎺柦] 鏃 %s.wav锛岃烦杩囨挱鎶", reason) + return + # 璁颁笅鍙 stop 鏃剁殑 stop_count锛屼綔涓"杩欐鏆傚仠褰掓垜"鐨勫嚟鎹 + # 婕忔枟鐨 stop 鎰忔濇槸"鍏堝埆璇达紝鎴戣鎻掓挱鎻愮ず"锛 + # 鐢ㄦ埛鐨 stop 鎰忔濇槸"闂槾锛屾垜涓嶆兂鍚" 鈥斺 涓よ呰蛋鍚屼竴鏉 8021 + # 鎺у埗鑵匡紝鏃犱粠鍖哄垎銆傛挱鎶ユ湡闂寸敤鎴疯嫢璇"鍋滀竴涓"锛 + # gate.stop() 浼氳 stop_count 鍐 +1锛涙鏃惰嫢杩樻棤鏉′欢 resume锛 + # 灏辨妸鐢ㄦ埛鐨勫仠姝㈢姸鎬佽鐩栨帀浜嗐 + # 锛堟挱鎶ユ敼鎴愬悗鍙颁换鍔′箣鍚庯紝杩欎釜骞跺彂绐楀彛鏇村ぇ浜嗐傦級 + _gate = getattr(manager, "gate", None) + _owned = getattr(_gate, "stop_count", None) if _gate else None + try: + await harness.send({"type": "funnel.stop", "reason": reason}) + LOG.info("[寮哄埗鎺柦] funnel.stop 宸插彂 (%s)", reason) + except Exception as e: + LOG.warning("[寮哄埗鎺柦] funnel.stop 澶辫触: %s", e) + await asyncio.sleep(0.35) # 绛 STOP 缁 8021鈫抮okid鈫抌lock_and_flush + try: + await speaker.resume() + await speaker.enqueue(wav_pcm, generation=0) + dur = len(wav_pcm) / 24000.0 + LOG.info("[寮哄埗鎺柦] 鎾 reject wav: %s (%.2fs)", reason, dur) + if live_rec is not None: + live_rec.log_event("HINT", f"{reason} ({dur:.2f}s)") + await asyncio.sleep(dur + 0.2) + except Exception as e: + LOG.warning("[寮哄埗鎺柦] 鎾 wav 澶辫触: %s", e) + # 鍙湁"杩欐鏆傚仠浠嶅綊鎴"鎵 resume銆俿top_count 鍙樹簡璇存槑鏈熼棿 + # 鏈夋柊鐨 stop 杩涙潵锛堢敤鎴锋寜鐨勶級锛屾鏃朵繚鎸佸仠姝㈢姸鎬佷笉鍔ㄣ + _now_cnt = getattr(_gate, "stop_count", None) if _gate else None + if _owned is not None and _now_cnt is not None and _now_cnt != _owned: + LOG.info("[寮哄埗鎺柦] 鎾姤鏈熼棿鏀跺埌鏂扮殑 STOP" + "锛坰top_count %s鈫%s锛夛紝淇濈暀鐢ㄦ埛鐨勫仠姝㈢姸鎬侊紝涓 resume", + _owned, _now_cnt) + return + try: + await harness.send({"type": "funnel.resume", "reason": "hint_done"}) + LOG.info("[寮哄埗鎺柦] funnel.resume 宸插彂锛堟彁绀烘挱瀹岋級") + except Exception as e: + LOG.warning("[寮哄埗鎺柦] funnel.resume 澶辫触: %s", e) + + # 鈹鈹 娑堣瀺鐢細閫 best 浣嗘案杩滄斁琛岋紙鍏抽棴宸ヤ綔鈶㈢殑鎷掔粷/鎾姤锛夆攢鈹 + # 鐢ㄩ旓細鎶"澶氬抚閫 best"(宸ヤ綔鈶)鐨勬敹鐩婁粠"鎷掔粷鍧忓浘"(宸ヤ綔鈶)閲屽墺鍑烘潵銆 + # 瀵圭収 bare锛堟瘡杞彇涓棿甯х洿鍙戯級锛屾湰 arm 姣忚疆鍙 best 鐩村彂锛 + # 涓嬫父妯″瀷渚у畬鍏ㄤ竴鑷达紝宸紓鍙潵鑷夊浘銆 + if no_reject: + _b = decision.best + if _b: + ts = now_ms() + latest_frame.set(_b, ts) + stats.image_count += 1 + await harness.send_frame(_b, latest_frame.sequence, ts) + _emit_chunk_img(web_ui, _b, latest_frame.sequence, True) + if live_rec is not None: + try: + await asyncio.to_thread( + live_rec.on_frame, _b, latest_frame.sequence) + except Exception as e: + LOG.warning("[LIVE] on_frame err: %s", e) + LOG.info("[婕忔枟-鏀捐] best (鍘熷垽瀹=%s)", decision.reason) + # 娑堣瀺 arm 涓嶆墽琛屾嫆缁濓紝浣**鍒ゅ埆缁撴灉瑕佺暀鐥**锛 + # 杩欎竴杞紡鏂楁湰鏉ヤ細涓嶄細鎷︺佹嫤鐨勭悊鐢辨槸浠涔堬紝鏄垎鏋愬垽鍒惧悜鐨勪緷鎹 + if live_rec is not None and not decision.send: + live_rec.log_event( + "WOULD-REJECT", + f"{decision.reason} {decision.hint or ''}") + else: + LOG.info("[婕忔枟-鏀捐] 鏈疆鏃 best锛%s锛夛紝璺宠繃", decision.reason) + elapsed = time.monotonic() - t0 + try: + await asyncio.wait_for(stop_evt.wait(), + timeout=max(0.0, interval_s - elapsed)) + except asyncio.TimeoutError: + pass + continue + + # 绗竴瑙嗚锛氭妸杩欎竴杞噰闆嗗埌鐨**鎵鏈夊抚**閮芥帹鍑哄幓锛屼笉绠℃斁娌℃斁琛屻 + # 鈽 蹇呴』鏀惧湪鎵鏈 reject 鍒嗘敮**涔嬪墠** 鈥斺 鎸佺画鍧忛偅鏉℃湯灏炬湁 continue锛 + # 鏀惧湪鍚庨潰鐨勮瘽锛屸懀 鎷掔粷瀵嗛泦鏃舵暣杞兘璺宠繃鎺ㄥ浘锛岀敾闈㈠氨瀹氫綇浜嗐 + # 鍙帹 best 鐨勮瘽锛屸懀 鎷﹀緱澶氭椂鐢婚潰浼氫竴鍗′竴鍗★紙瀹炴祴灏辨槸杩欎釜鐜拌薄锛夛紱 + # 鑰屼笖鐪嬩笉鍒"琚嫤鐨勫浘闀夸粈涔堟牱"銆傛暣绨囬兘鎺紝鐢婚潰杩炵画锛 + # 鍓嶇杩樿兘鎸 img_sent / reject_reason 鏍囧嚭鍝紶鐪熼佷簡妯″瀷銆 + if web_ui is not None and getattr(web_ui, "live_clients", None): + _all = list(decision.frames or []) + _bi = decision.best_index if isinstance(decision.best_index, int) else -1 + if not _all and decision.best: + _all, _bi = [decision.best], 0 + for _i, _f in enumerate(_all): + _is_best_sent = bool(decision.send) and _i == _bi + _emit_chunk_img(web_ui, _f, latest_frame.sequence, + _is_best_sent, + reason="" if decision.send else (decision.reason or "")) + + if decision.send and decision.best: + ts = now_ms() + latest_frame.set(decision.best, ts) + stats.image_count += 1 + await harness.send_frame(decision.best, latest_frame.sequence, ts) + LOG.info("[婕忔枟] send (%s)", decision.reason) + # -o record锛氬彧褰曠湡鍙戦佺粰妯″瀷鐨 best 鍥撅紙chunk_idx = sequence锛 + if live_rec is not None: + try: + await asyncio.to_thread( + live_rec.on_frame, decision.best, latest_frame.sequence) + except Exception as e: + LOG.warning("[LIVE] on_frame err: %s", e) + # 娉ㄦ剰锛歴end 涓嶆竻 _pending_reject锛屼繚鎶ゆ湡鍙湅鍒版爣鐐归偅涓鍒 + else: + LOG.info("[婕忔枟] reject(%s) -> hint: %s", + decision.reason, decision.hint) + if live_rec is not None: + live_rec.log_event("REJECT", f"{decision.reason} {decision.hint or ''}") + if force_measure and speaker is not None and decision.reason != "need_focus": + now_mono = time.monotonic() + if _pending_reject is not None: + pass # 宸插湪淇濇姢涓紝涓婇潰宸插鐞嗭紝涓嶉噸澶嶆寕璧 + elif sent_tracker is not None and sent_tracker.speaking: + # 妯″瀷姝e康瀛楋細鎸傝捣锛屼繚鎶ゅ埌涓嬩竴涓爣鐐瑰啀澶嶆煡 + _pending_reject = { + "since": now_mono, + "reason": decision.reason, + "snap": sent_tracker.snapshot(), + } + LOG.info("[鏍囩偣淇濇姢] speak涓紝reject(%s)鎸傝捣锛岀瓑璇煶鎾埌鏍囩偣鎴栬秴鏃1.5s", + decision.reason) + else: + # 妯″瀷娌″康瀛楋紙listen/闈欓粯锛夆啋 鏃犻渶淇濇姢锛岀洿鎺ユ墦鏂紙鍙楄妭娴侊級 + if (not _hint_playing + and now_mono - _last_hint_mono >= _HINT_THROTTLE_S): + _last_hint_mono = now_mono + asyncio.create_task(_do_reject_interrupt(decision.reason)) + # 鈹鈹 銆屾寔缁潖銆嶆娴嬶紙浼樺厛浜庢爣鐐逛繚鎶わ級鈹鈹 + # 3s 绐楀彛鍐"鐪熷潖"reject鈮2 鈫 杈撳嚭鍩轰簬鍧忓浘=骞昏銆傛鏃朵笉璧版爣鐐逛繚鎶 + # 锛堝墠鍗婁篃閿欙紝涓嶅煎緱淇濇姢锛夛紝鐩存帴瑙﹀彂"鍋溾啋鎾彁绀衡啋鎭㈠"銆 + # need_focus 涓嶇畻鍧忥細瀹冩槸绯荤粺姝e湪瀵圭劍(run_once 宸插彂 /reg 閲嶆姄)锛 + # 灞炰簬鍐呴儴鑷剤锛屼笉鏄敤鎴烽犳垚鐨勬寔缁潖锛岀畻杩涘幓浼氳鍒ゃ佸共鎵板鐒︽祦绋嬨 + # "鐪熷潖" = reject 涓 reason 涓嶆槸 need_focus锛坰evere_shake/unstable/ + # too_dark/orient 绛夛紝瑕佺敤鎴峰姩鎵嬬殑锛夈 + _now = time.monotonic() + _is_real_bad = (not decision.send) and (decision.reason != "need_focus") + _judge_history.append((_now, _is_real_bad)) + _judge_history[:] = [(t, r) for (t, r) in _judge_history + if _now - t <= _WINDOW_S] + _bad_in_window = sum(1 for (_t, r) in _judge_history if r) + + if _bad_in_window >= _BAD_THRESH and _is_real_bad: + _pending_reject = None # 鎸佺画鍧忎紭鍏堬紝鍙栨秷鏍囩偣淇濇姢鎸傝捣 + # 鍒拌繖閲屽繀鏄"鐪熷潖"(severe_shake/unstable/too_dark/orient)銆 + # **蹇呴』鍜屾爣鐐逛繚鎶ら偅鏉′竴鏍峰彈 _HINT_THROTTLE_S 绾︽潫**锛 + # 鍘熸敞閲婅"鑰楁椂鈮堟挱鎶ユ椂闀匡紝澶╃劧闂撮殧"锛屼絾閭d笉鎴愮珛 鈥斺 + # STOP/RESUME 鏄湪鎾姤**涔嬪墠**灏卞彂鍑哄幓鐨勶紝涓嶅彈鎾姤鏃堕暱闄愬埗銆 + # 瀹炴祴鐢婚潰鎸佺画涓嶅悎鏍兼椂锛堟檭鍔/鍊掔疆锛夛紝3s 绐楀彛姣忚疆閮借兘鍑戝 2 娆$湡鍧忥紝 + # 浜庢槸姣忚疆涓娆 STOP/RESUME锛17 绉掑唴 17 涓 control event 涔嬪悗 + # gateway 浼氳瘽鐩存帴鎹唬閲嶈繛锛坓eneration 0鈫1锛夛紝鏈熼棿 + # gateway=preparing銆侀煶棰戦槦鍒楀牭姝 queue=96 drops 鏆存定銆佸浘涔熼佷笉鍑哄幓銆 + # 琛ㄧ幇灏辨槸"鐏彉绾€佺敾闈㈠崱浣忋佸彇鍥 TimeoutError"銆 + if _hint_playing or _now - _last_hint_mono < _HINT_THROTTLE_S: + LOG.debug("[鎸佺画鍧廬 %d娆$湡鍧忥紝浣嗚窛涓婃鎻愮ず %.1fs < %.1fs锛岃烦杩囨墦鏂", + _bad_in_window, _now - _last_hint_mono, + _HINT_THROTTLE_S) + elapsed = time.monotonic() - t0 + try: + await asyncio.wait_for(stop_evt.wait(), + timeout=max(0.0, interval_s - elapsed)) + except asyncio.TimeoutError: + pass + continue + _last_hint_mono = _now + LOG.info("[鎸佺画鍧廬 3s鍐%d娆$湡鍧弐eject锛屾墦鏂+鎾彁绀猴紙鍋溾啋鎾啋鎭㈠锛", + _bad_in_window) + if live_rec is not None: + live_rec.log_event( + "PERSIST-BAD", + f"3s鍐厈_bad_in_window}娆$湡鍧 鈫 鎵撴柇 ({decision.reason})") + asyncio.create_task(_do_reject_interrupt(decision.reason)) + elapsed = time.monotonic() - t0 + try: + await asyncio.wait_for(stop_evt.wait(), + timeout=max(0.0, interval_s - elapsed)) + except asyncio.TimeoutError: + pass + continue + + # 鈹鈹 銆屾爣鐐逛繚鎶ゃ嶆牳蹇冿細绐佸彂 reject 涓嶇珛鍗虫墦鏂紝淇濇姢褰撳墠娈佃闊虫挱鏀惧埌 + # 涓嬩竴涓彞鏈爣鐐癸紝鍒版爣鐐归偅涓鍒诲鏌ュ綋鏃跺垽瀹氾紱浠 reject 鎵嶆墦鏂紝鍚﹀垯鏃犱簨銆 + # 鎸傝捣鏈熼棿鐨 send/reject 閮戒笉绠楁暟锛屽彧鐪"鍒版爣鐐归偅涓鍒"鐨勫垽瀹氥 + # 锛坰napshot 璁版爣鐐瑰簭鍙凤紝boundary_reached 鍒ゆ柇鏂版爣鐐+鍏惰闊冲凡鎾畬锛 + if force_measure and speaker is not None and _pending_reject is not None: + # 姝e湪淇濇姢涓細鍙垽鏂"鏄惁鍒版爣鐐(璇煶鎾畬)鎴栬秴鏃"锛屽埌浜嗘墠鐢ㄥ綋鏃跺垽瀹氬鏌 + now_mono = time.monotonic() + reached = (sent_tracker is not None + and sent_tracker.boundary_reached( + _pending_reject["snap"], speaker)) + timed_out = (now_mono - _pending_reject["since"]) >= 1.5 + if reached or timed_out: + # 鍒版爣鐐/瓒呮椂 鈫 澶嶆煡姝ゅ埢鍒ゅ畾 + if decision.send: + LOG.info("[鏍囩偣淇濇姢] %s锛屾鍒诲凡鎭㈠(send)锛屾棤浜嬪彂鐢", + "鍒版爣鐐" if reached else "瓒呮椂1.5s") + _pending_reject = None + elif decision.reason == "need_focus": + _pending_reject = None # 澶嶆煡鏄鐒︼紝闈欓粯 + else: + # 浠 reject 鈫 鎵撴柇锛堝彈鑺傛祦绾︽潫锛 + LOG.info("[鏍囩偣淇濇姢] %s锛屼粛 reject(%s)锛屾墦鏂", + "鍒版爣鐐" if reached else "瓒呮椂1.5s", decision.reason) + if (not _hint_playing + and now_mono - _last_hint_mono >= _HINT_THROTTLE_S): + _last_hint_mono = now_mono + _pending_reject = None + asyncio.create_task(_do_reject_interrupt(decision.reason)) + else: + _pending_reject = None # 琚妭娴 + + if probe is not None: + since_last = (t0 - _last_grab_mono) * 1000.0 if _last_grab_mono else 0.0 + jb = len(decision.best) if decision.best else 0 + tm = decision.timings + probe.mark_grab( + latest_frame.sequence, grab_ms, jb, since_last, + reason=decision.reason, + judge_ms=tm.get("judge_ms"), + af_ms=tm.get("af_ms"), + n_frames=tm.get("n_frames"), + ) + _last_grab_mono = t0 + except Exception as e: + LOG.warning("[ESP32] image loop error: %s", e, exc_info=True) + # 鎺у埗鍙栧浘鑺傚 + elapsed = time.monotonic() - t0 + try: + await asyncio.wait_for(stop_evt.wait(), timeout=max(0.0, interval_s - elapsed)) + except asyncio.TimeoutError: + pass + LOG.info("[ESP32] image loop stopped") + + +# ============================================================ +# ESP32 Runtime锛氱户鎵 rokid 鐨 PhaseBRokidRuntime锛屽鐢ㄥ叏閮 harness 楠ㄦ灦锛 +# 鍙鐩 start()锛氫笉璧 web server锛屾敼璧蜂袱涓 ESP32 涓诲姩鎷夊彇 task銆 +# ============================================================ +from .rokid_runtime import PhaseBRokidRuntime + + +def _insecure_ssl_for_wss(url: str) -> Optional[ssl.SSLContext]: + """wss:// 涓旇嚜绛惧悕璇佷功鏃讹紝杩斿洖涓涓笉鏍¢獙璇佷功鐨 ssl context锛泈s:// 杩斿洖 None銆 + + 涓 rokid GatewayDuplexSession._ssl_context 鍚屾鍋氭硶锛坈heck_hostname=False, + verify_mode=CERT_NONE锛夛紝鐢ㄤ簬杩炴湰鍦拌嚜绛惧悕鐨 8021銆""" + if not url.lower().startswith("wss://"): + return None + ctx = ssl.create_default_context() + ctx.check_hostname = False + ctx.verify_mode = ssl.CERT_NONE + return ctx + + +class TlsHarnessClient(HarnessClient): + """涓 HarnessClient 瀹屽叏涓鑷达紝浠呭湪 ws_connect 鏃跺 wss:// 浼犲叆鑷鍚 ssl銆 + + 瑕嗙洊 run()锛氶愯鐓ф妱鐖剁被锛屽敮涓鍖哄埆鏄 ws_connect 澶氫簡 ssl= 鍙傛暟銆""" + + async def run(self) -> None: + ssl_ctx = _insecure_ssl_for_wss(self.url) + while not self._stop.is_set(): + try: + async with aiohttp.ClientSession() as client: + async with client.ws_connect( + self.url, heartbeat=30, max_msg_size=0, ssl=ssl_ctx + ) as ws: + self.ws = ws + self.connected = True + LOG.info("Harness connected: %s", self.url) + async for message in ws: + if message.type != aiohttp.WSMsgType.TEXT: + if message.type in ( + aiohttp.WSMsgType.CLOSED, + aiohttp.WSMsgType.ERROR, + ): + break + continue + payload = json.loads(message.data) + message_type = payload.get("type") + if message_type == "harness.ready": + await self.manager.emit_recovery_sync() + elif message_type == "control.intent": + task = asyncio.create_task( + self.manager.handle_control(payload) + ) + self._control_tasks.add(task) + task.add_done_callback(self._control_tasks.discard) + except asyncio.CancelledError: + raise + except Exception as exc: + if not self._stop.is_set(): + LOG.warning("Harness connection failed: %s", exc) + finally: + self.connected = False + self.ws = None + if not self._stop.is_set(): + await asyncio.sleep(self.reconnect_s) + + +class PhaseBEsp32Runtime(PhaseBRokidRuntime): + """ESP32 鐗 runtime锛欼/O 浠 'PC 璧 server 琚姩鏀' 鎹㈡垚 'PC 涓诲姩鎷 ESP32'銆 + + __init__ / close / health / _initial_session_loop / _stats_loop 鍏ㄩ儴缁ф壙銆 + 浠呰鐩 start()锛氭妸 Rokid 鐨 web server 鎹㈡垚 esp32_audio_reader + esp32_image_loop銆 + """ + + def __init__(self, config, esp32_host, esp32_port, image_tcp_port, + image_interval_s, image_timeout_s, probe=None, + funnel=None, recorder=None, force_measure=False, + gateway_host="127.0.0.1", gateway_port=8040, + reject_wav_dir="assets/reject_wav", + live_record_dir=None, rerun_from=None, + funnel_rerun_from=None, web_ui_port=None, web_ui_host="127.0.0.1", + record_no_media: bool = False, + rerun_tail_wait_s: float = 30.0, + no_reject: bool = False, + rotate: int = 0, **kwargs): + super().__init__(config, **kwargs) + # 鐢ㄦ敮鎸佽嚜绛惧悕 wss 鐨 harness client 鏇挎崲鐖剁被寤哄ソ鐨勬櫘閫 HarnessClient锛 + # 骞跺悓姝ユ洿鏂 manager.harness 寮曠敤锛堝彂閬ユ祴/闀滃儚璧板悓涓涓級銆 + self.harness = TlsHarnessClient( + config.harness_url, + config.harness_client_id, + self.manager, + config.reconnect_s, + ) + self.manager.harness = self.harness + self._esp32_host = esp32_host + self._esp32_port = esp32_port + self._image_tcp_port = image_tcp_port + self._image_interval_s = image_interval_s + self._image_timeout_s = image_timeout_s + self._tcp_img = TcpImageClient(esp32_host, image_tcp_port, + rotate=int(rotate) % 360) + self._stop_evt = asyncio.Event() + self._probe = probe + self._funnel = funnel + self._recorder = recorder + self._force_measure = force_measure + self._reject_wav_dir = reject_wav_dir + self._gateway_host = gateway_host + self._gateway_port = gateway_port + # 蹇垫彁绀鸿蛋 gateway wss /ws/chat锛岃嚜绛惧悕璇佷功 鈫 澶嶇敤鍚屾 insecure ssl + self._chat_ssl = _insecure_ssl_for_wss(f"wss://{gateway_host}:{gateway_port}/ws/chat") + + self._rerun_from = rerun_from + self._funnel_rerun_from = funnel_rerun_from + self._rerun_tail_wait_s = rerun_tail_wait_s + self._no_reject = no_reject + self._rotate = int(rotate) % 360 + self._replay_t0 = None # rerun 缁熶竴鍥炴斁闆剁偣锛堝氨缁椂璁撅級 + self._rec_client = None + # 鍥炴斁涓ゆ潯娴佸悇鑷殑瀹屾垚鏍囧織锛氬浘杞暟甯稿皯浜庨煶棰戠鏁帮紝璋佸厛瀹岄兘涓嶈兘鎺愭瀵规柟 + self._audio_done = asyncio.Event() + self._image_done = asyncio.Event() + # bridge_ui 绗竴瑙嗚锛8080锛夈俵ive/rerun/funnel-rerun 閮藉彲鎺 img_b64銆 + self._web_ui = None + if web_ui_port and WebUIServer is not None: + try: + self._web_ui = WebUIServer( + port=web_ui_port, + host=web_ui_host, + sessions_root=Path(live_record_dir).parent if live_record_dir + else Path("live_sessions"), + # 鎺ヤ笂杩愯鏃剁殑鍋滄浜嬩欢锛氫笉鎺ョ殑璇 POST /api/stop 浼氳繑鍥炴垚鍔熴 + # 鍓嶇鎸夐挳涔熷彉鐏帮紝浣嗚繍琛屾椂鏍规湰娌″仠锛宻ession 涔熶笉浼氭敹灏捐惤鐩樸 + stop_callback=self._stop_evt.set, + mode_info={"mode": "rerun" if (rerun_from or funnel_rerun_from) + else "live"}, + ) + except Exception as e: + LOG.warning("[UI] WebUIServer 鍒涘缓澶辫触: %s", e) + self._web_ui = None + # -o record锛堥煶+鍥撅紝鍙瓨鐪熷彂閫佺殑 best锛夛細鍜屾紡鏂 record 骞跺瓨銆 + self.live_rec = (LiveRecorder(live_record_dir, no_media=record_no_media) + if live_record_dir else None) + if self.live_rec is not None: + try: + self.live_rec.attach_to_player(self.speaker) # 褰 AI 闊 + except Exception as e: + LOG.warning("[LIVE] attach_to_player 澶辫触: %s", e) + + # 鍙ュ瓙杈圭晫杩借釜鍣細鍖呰 manager.handle_result锛屾瘡涓ā鍨 chunk 鏇存柊鍙ュ瓙/璇煶杩涘害锛 + # 渚 image_loop 鐨"鏍囩偣淇濇姢"绛栫暐璇诲彇锛坮eject 鏃朵繚鎶ゅ綋鍓嶆鍒颁笅涓涓爣鐐瑰啀澶嶆煡锛夈 + self.sent_tracker = SentenceTracker() + self._turn_printer = TurnPrinter() + self._turn_start_ms = None # 褰撳墠 turn 鐨勮捣鐐癸紙session 鍐呯浉瀵 ms锛 + _orig_handle_result = self.manager.handle_result + + async def _wrapped_handle_result(session, result): + try: + is_listen = bool(result.get("is_listen")) + text = str(result.get("text") or "") + end_of_turn = bool(result.get("end_of_turn")) + ab64 = str(result.get("audio_data") or "") + audio_ms = 0.0 + if ab64 and not is_listen: + n = len(base64.b64decode(ab64)) // 4 # float32 + audio_ms = n * 1000.0 / 24000.0 + self.sent_tracker.update(text, audio_ms, is_listen) + + # 鈹鈹 璇勫垎鐢細鎶 chunk 纰庣墖鑱氬悎鎴愭暣娈 turn锛岃惤 transcript/subtitles 鈹鈹 + if self.live_rec is not None: + # 鍏堣鍘熷閫 chunk锛堜笉渚濊禆 end_of_turn锛屼繚璇佷竴瀹氭湁涓滆タ鍙瘎鍒嗭級 + self.live_rec.log_model_chunk(text, is_listen, end_of_turn, audio_ms) + prev_idx = self._turn_printer.turn_idx + turn_idx_cur, full_text, turn_listen = self._turn_printer.feed( + is_listen, end_of_turn, text) + if text and self._turn_printer.turn_idx > prev_idx: + self._turn_start_ms = self.live_rec.session_ms() + # 鍙鎷垮埌瀹屾暣娈靛氨钀界洏锛坋nd_of_turn 鎴 listen/speak 鍒囨崲鍐插嚭鐨勶級 + if full_text: + end_ms = self.live_rec.session_ms() + 800 # 澶氭寕 0.8s + start_ms = (self._turn_start_ms + if self._turn_start_ms is not None + else max(end_ms - 3000, 0)) + self.live_rec.log_turn_text(turn_idx_cur, full_text, turn_listen) + self.live_rec.log_subtitle( + start_ms=start_ms, end_ms=end_ms, text=full_text, + is_listen=turn_listen, turn_idx=turn_idx_cur) + self._turn_start_ms = None + + # 鎺ㄦā鍨嬫枃瀛楀埌 bridge_ui 绗竴瑙嗚锛坱ype=result锛 + if self._web_ui is not None and getattr(self._web_ui, "live_clients", None): + asyncio.create_task(self._web_ui.emit({ + "type": "result", + "is_listen": is_listen, + "end_of_turn": end_of_turn, + "text": text, + })) + except Exception: + pass + return await _orig_handle_result(session, result) + + self.manager.handle_result = _wrapped_handle_result + + async def _gate_open_when_ready(self, ready_evt: asyncio.Event) -> None: + """绛 gateway 灏辩华鍚庢斁琛屽洖鏀句换鍔°""" + try: + await self._wait_gateway_ready() + finally: + # 褰曞埗鐨 t0 蹇呴』鍜屽洖鏀捐捣鐐逛竴鑷达紝鍚﹀垯 frames.jsonl/subtitles 鐨勬椂闂磋酱 + # 浼氭妸鎺掗槦閭e嚑鍗佺涔熺畻杩涘幓锛屼袱涓 arm 鏃犳硶瀵归綈銆 + if self.live_rec is not None: + self.live_rec.start() + LOG.info("[LIVE] rerun 褰曞埗宸插紑锛堝氨缁悗鍚姩锛岀粨鏉熻嚜鍔ㄥ嚭 mp4锛") + # 缁熶竴鍥炴斁闆剁偣锛氶煶棰戞寜 40ms 缁濆鏃堕挓鎺ㄣ佸浘鎸 frames.jsonl 鐨 t 绛夊緟锛 + # 涓よ呭繀椤荤敤鍚屼竴涓 t0锛屽惁鍒欏悇鑷互"鑷繁琚皟搴﹀埌鐨勯偅涓鍒"涓洪浂鐐癸紝 + # 璧疯窇宸灏戝叏鐪嬩簨浠跺惊鐜紝闊冲浘灏卞涓嶉綈銆 + self._replay_t0 = time.monotonic() + _rc = getattr(self, "_rec_client", None) + if _rc is not None: + _rc.replay_t0 = self._replay_t0 + ready_evt.set() + + async def _wait_gateway_ready(self, timeout_s: float = 180.0) -> bool: + """绛 duplex session 鐪熸 prepared 涔嬪悗鍐嶅紑濮嬪洖鏀俱 + + 瀹炴祴锛氳繛鐫璺戜袱娆 rerun 鏃讹紝鍚庝竴娆′細鍦 gateway 鎺掗槦锛圼GW] queue position=1锛 + 闀胯揪 17 绉掓墠 prepared銆傝屽洖鏀句换鍔′粠绗 0 绉掑氨鎺ㄩ煶棰戝拰鍥 鈥斺 杩 17 绉掔殑杈撳叆 + 鍏ㄩ儴鎺ㄧ粰涓涓繕涓嶅瓨鍦ㄧ殑 session锛岀洿鎺ヤ涪鎺夛紝鎻愰棶灏卞湪閲岄潰锛屾ā鍨嬭嚜鐒舵病鍙嶅簲銆 + 鏇磋鍛界殑鏄袱涓 arm 琚悶鎺夌殑闀垮害涓嶄竴鏍凤紝閰嶅璁捐鐩存帴澶辨晥锛岃屼笖浜嬪悗浠庣粨鏋滀笂 + 鐪嬩笉鍑烘潵锛堥暱寰楀氨鍍"妯″瀷娌″洖绛"锛夈傛墍浠ュ洖鏀惧墠蹇呴』绛夊氨缁 + """ + t0 = time.monotonic() + warned = False + while time.monotonic() - t0 < timeout_s: + if self._stop_evt.is_set(): + return False + try: + status = str(self.manager.health().get("gateway_status") or "") + except Exception: + status = "" + if status == "running": + waited = time.monotonic() - t0 + if waited > 1.0: + LOG.info("[RERUN] gateway 灏辩华锛堢瓑浜 %.1fs锛夛紝寮濮嬪洖鏀", waited) + return True + if not warned and time.monotonic() - t0 > 3.0: + warned = True + LOG.info("[RERUN] 绛 gateway session 灏辩华鈥︼紙status=%s锛", status or "?") + await asyncio.sleep(0.25) + LOG.warning("[RERUN] 绛 gateway 灏辩华瓒呮椂 %.0fs锛屼粛寮濮嬪洖鏀撅紙鏈缁撴灉鍙兘涓嶅彲鐢級", + timeout_s) + return False + + async def start(self) -> None: + await self.speaker.start() + _install_asr_tap(self.harness, self.live_rec) + if self._web_ui is not None: + try: + await self._web_ui.start() + LOG.info("[UI] bridge_ui 绗竴瑙嗚宸插惎鍔: http://localhost:%d", + self._web_ui.port) + except Exception as e: + LOG.warning("[UI] bridge_ui 鍚姩澶辫触: %s", e) + self._web_ui = None + # 鏂瑰悜鍒嗙被鍣ㄩ鐑紙鍘熻璁★紝-o 杩佺Щ鏃舵紡浜嗭級锛氶娆 process_orientation 鍚ā鍨嬪姞杞+ + # oneDNN 缂栬瘧锛垀2s锛岀敋鑷 5s+锛夈備笉棰勭儹鐨勮瘽瀹冧細鎺ㄨ繜鍒扮涓娆 accept 鎵嶇幇鍦哄姞杞斤紝 + # 鍗′綇閭d竴杞紝涓斿湪姝や箣鍓嶉摼璺嚭涓嶄簡 send锛堟ā鍨嬮暱鏃堕棿鎷夸笉鍒板浘锛夈 + if self._funnel is not None: + # 鈽 鍏堝湪**涓荤嚎绋**閲屾妸 paddle/paddleocr 瀵艰繘鏉ャ + # cam_pipeline_v2 鐨 `from paddleocr import ...` 鏄噿鍔犺浇銆佸啓鍦ㄥ嚱鏁颁綋閲岋紝 + # 鑰 warmup_orient 璧 asyncio.to_thread锛堝伐浣滅嚎绋嬶級鈥斺 paddle 杩欑被 + # 甯﹀ぇ閲 C 鎵╁睍鍜屽唴閮ㄥ惊鐜緷璧栫殑鍖咃紝鍦ㄩ潪涓荤嚎绋嬮娆″鍏ユ椂浼氭挒涓 + # "partially initialized module 'paddle' has no attribute 'tensor' + # (most likely due to a circular import)"锛 + # 鐒跺悗 _ORI_CLS_TRIED 姘镐箙缃綅銆佹暣涓繘绋嬮鍥為敊璇殑鏃╂湡 CV 鍒ゆ嵁锛 + # 琛ㄧ幇灏辨槸绗竴杞氨璇姤 orient_flipped锛堝疄娴 panel 鍚姩蹇呯幇锛夈 + # 鍦ㄤ富绾跨▼棰勫涓娆★紝鍚庨潰绾跨▼閲屽啀 import 灏辨槸鎷跨紦瀛橈紝涓嶄細鍐嶈Е鍙戙 + try: + # 鈽 鍙杩欎竴涓鍙凤紝**涓嶈 `import paddleocr` 鎴 `import paddle`**銆 + # `import paddleocr` 浼氳蛋 paddlex.inference.utils.official_models + # 鈫 import modelscope 鈫 modelscope.utils.logger 鈫 import torch锛 + # 鑰岃繖涓幆澧冮噷 torch 鐨 shm.dll 鍔犺浇澶辫触锛圵inError 127锛夛紝 + # 浜庢槸鏁存潯閾炬柇鎺夈佹柟鍚戝垎绫诲櫒閫鍥為敊璇殑鏃╂湡 CV 鍒ゆ嵁銆 + # 绐勫鍏ヤ笉缁忚繃閭d竴鏀紝瀹炴祴鍙敤锛坈am_pipeline_v2 閲屼篃鏄繖涔堝啓鐨勶級銆 + from paddleocr import DocImgOrientationClassification # noqa: F401 + LOG.info("[棰勭儹] paddleocr 鏂瑰悜鍒嗙被鍣ㄥ凡鍦ㄤ富绾跨▼棰勫鍏") + except Exception as e: + LOG.warning("[棰勭儹] 鏂瑰悜鍒嗙被鍣ㄩ瀵煎叆澶辫触锛堝彲鑳介鍥炶交閲忓垽鎹級: %s", e) + try: + ms = await asyncio.to_thread(self._funnel.warmup_orient) + # 蹇呴』鏌ョ湡瀹炵姸鎬侊細杩欒鍘熸潵鏃犳潯浠舵墦"灏辩华"锛岃屽姞杞藉け璐ユ椂 + # process_orientation 浼氶潤榛樺洖閫鍒拌交閲 CV 鍒ゆ嵁銆佺収鏍疯繑鍥烇紝 + # 浜庢槸"棰勭儹灏辩华"鏍规湰涓嶈兘璇佹槑鍒嗙被鍣ㄥ彲鐢紙瀹炴祴 panel 鐜涓 + # paddle 寰幆瀵煎叆澶辫触 鈫 杞婚噺鍒ゆ嵁 鈫 绗竴杞氨璇垽 orient_flipped锛夈 + try: + from .cam_pipeline_v2 import orient_classifier_status + _ok, _err = orient_classifier_status() + except Exception: + _ok, _err = True, "" + if _ok: + LOG.info("[棰勭儹] 鏂瑰悜鍒嗙被鍣ㄥ氨缁 (棣栨 %.0fms锛岃繍琛屾椂搴旈檷鍒 ~10ms)", ms) + else: + LOG.warning( + "[棰勭儹] !! PaddleOCR 鏂瑰悜鍒嗙被鍣ㄥ姞杞藉け璐ワ紝宸插洖閫鍒拌交閲 CV 鍒ゆ嵁 鈥斺 " + "orient 鍒ゅ畾浼氭槑鏄惧彉宸紙瀹炴祴绗竴杞氨璇姤 flipped锛夈傚師鍥: %s", _err) + except Exception as e: + LOG.warning("[棰勭儹] 鏂瑰悜鍒嗙被鍣ㄩ鐑烦杩: %s", e) + # 鈹鈹 rerun 妯″紡锛氱敤褰曞埗鐨 session 鍥炴斁锛堥煶棰+best鍥撅級閲嶈窇 -o锛屼笉鎺ュ疄鏃惰澶 鈹鈹 + if self._rerun_from: + LOG.info("[RERUN] 妯″紡鍚姩锛屽洖鏀 session: %s", self._rerun_from) + _ready = asyncio.Event() + asyncio.create_task(self._gate_open_when_ready(_ready)) + self._tasks = [ + asyncio.create_task(self.harness.run()), + asyncio.create_task(self._initial_session_loop()), + asyncio.create_task(self._stats_loop()), + asyncio.create_task(rerun_audio_reader( + self._rerun_from, + self.manager, self.harness, + self.audio_queue, self.audio_mirror, self.stats, + self.config.input_gain, self._stop_evt, + live_rec=self.live_rec, ready_evt=_ready, + replay_t0=lambda: self._replay_t0, + done_evt=self._audio_done, + )), + asyncio.create_task(rerun_image_loop( + self._rerun_from, self.latest_frame, self.harness, + self.stats, self._image_interval_s, self._stop_evt, + web_ui=self._web_ui, + live_rec=self.live_rec, ready_evt=_ready, + replay_t0=lambda: self._replay_t0, + done_evt=self._image_done, + peer_done=self._audio_done, + speaker=self.speaker, sent_tracker=self.sent_tracker, + tail_wait_s=self._rerun_tail_wait_s, + )), + ] + return + # 鈹鈹 funnel rerun锛氬綍鍒剁殑鏁寸皣澶氬抚閲嶈窇婕忔枟锛涗笉鍔 --funnel 鍒欎负鏃犳紡鏂楀鐓х粍 鈹鈹 + if self._funnel_rerun_from: + no_funnel = (self._funnel is None) + _ready = asyncio.Event() + asyncio.create_task(self._gate_open_when_ready(_ready)) + LOG.info("[RERUN] %s 妯″紡鍚姩: %s", + "鏃犳紡鏂楀鐓х粍" if no_funnel else "funnel 閲嶈窇婕忔枟+鎾姤", + self._funnel_rerun_from) + n_frames = getattr(self._funnel, "n_frames", 3) if self._funnel else 3 + rec_client = RecordedImageClient(self._funnel_rerun_from, n_frames, + no_funnel=no_funnel) + self._rec_client = rec_client # 灏辩华鏃舵敞鍏ョ粺涓闆剁偣 + self._tasks = [ + asyncio.create_task(self.harness.run()), + asyncio.create_task(self._initial_session_loop()), + asyncio.create_task(self._stats_loop()), + asyncio.create_task(rerun_audio_reader( + self._funnel_rerun_from, + self.manager, self.harness, + self.audio_queue, self.audio_mirror, self.stats, + self.config.input_gain, self._stop_evt, + live_rec=self.live_rec, ready_evt=_ready, + replay_t0=lambda: self._replay_t0, + done_evt=self._audio_done, + )), + # 澶嶇敤瀹屾暣 esp32_image_loop锛坮un_once 閲嶅垽 + reject 鍋滄挱鎭㈠ + 鏍囩偣淇濇姢锛夛紝 + # 鍙妸鍥炬簮浠 TCP 鎹㈡垚 RecordedImageClient锛堣褰曞埗鏁寸皣澶氬抚锛夈 + asyncio.create_task(esp32_image_loop( + rec_client, self.latest_frame, self.harness, self.stats, + self._image_interval_s, self._image_timeout_s, self._stop_evt, + probe=self._probe, funnel=self._funnel, recorder=None, + speaker=self.speaker, + gateway_host=self._gateway_host, gateway_port=self._gateway_port, + ssl_ctx=self._chat_ssl, force_measure=self._force_measure, + manager=self.manager, + reject_wav_dir=self._reject_wav_dir, + sent_tracker=self.sent_tracker, + live_rec=self.live_rec, # 寮 --record-live 鍒欏綍 rerun 杈撳嚭鈫掕嚜鍔ㄥ嚭 mp4 + web_ui=self._web_ui, ready_evt=_ready, + tail_wait_s=self._rerun_tail_wait_s, + no_reject=self._no_reject, + peer_done=self._audio_done, done_evt=self._image_done, + )), + ] + return + # 璁 ESP32 鍒嗚鲸鐜囦负 HD(1280脳720)銆傚浐浠堕粯璁 SVGA(800脳600)锛屼絾婕忔枟闃堝兼槸鎸 HD + # 鏍囧畾鐨勶紙cam_pipeline_v2 娉ㄩ噴锛氬垎杈ㄧ巼鍥哄畾 HD 鍚庨拡瀵 HD 鏍囧畾锛夈-o 杩佺Щ鏃舵紡浜嗚繖姝ワ紝 + # 瀵艰嚧涓鐩磋窇 SVGA銆佺敾璐ㄤ笌闃堝间笉鍖归厤銆傝繖閲屽惎鍔ㄦ椂琛ヤ笂銆 + if self._funnel is not None: + ok = await asyncio.to_thread(self._funnel.set_resolution_hd) + LOG.info("[ESP32] 璁惧垎杈ㄧ巼 1280脳720(UXGA妗): %s", + "鎴愬姛" if ok else "澶辫触(妫鏌SP32 /control)") + await asyncio.sleep(0.3) # 鍒囧垎杈ㄧ巼鍚庡浐浠堕噸閰嶏紝绋嶇瓑 + if self.live_rec is not None: + self.live_rec.start() + LOG.info("[LIVE] -o record started") + self._tasks = [ + # 鈹鈹 缁ф壙鑷 rokid 鐨勪笁涓鏋 task 鈹鈹 + asyncio.create_task(self.harness.run()), + asyncio.create_task(self._initial_session_loop()), + asyncio.create_task(self._stats_loop()), + # 鈹鈹 ESP32 鐗规湁锛氫富鍔ㄦ媺鍙栭煶瑙嗛 鈹鈹 + asyncio.create_task(esp32_audio_reader( + self._esp32_host, self._esp32_port, + self.manager, self.harness, + self.audio_queue, self.audio_mirror, self.stats, + self.config.input_gain, self._stop_evt, + probe=self._probe, + live_rec=self.live_rec, + )), + asyncio.create_task(esp32_image_loop( + self._tcp_img, self.latest_frame, self.harness, self.stats, + self._image_interval_s, self._image_timeout_s, self._stop_evt, + probe=self._probe, funnel=self._funnel, recorder=self._recorder, + speaker=self.speaker, + gateway_host=self._gateway_host, gateway_port=self._gateway_port, + ssl_ctx=self._chat_ssl, force_measure=self._force_measure, + manager=self.manager, + reject_wav_dir=self._reject_wav_dir, + sent_tracker=self.sent_tracker, + live_rec=self.live_rec, + web_ui=self._web_ui, + no_reject=self._no_reject, + )), + ] + + async def close(self) -> None: + self._stop_evt.set() + if self._web_ui is not None: + try: + await self._web_ui.stop() + except Exception: + pass + if self.live_rec is not None: + try: + # 鍐叉帀杩樻病绛夊埌 end_of_turn 鐨勯偅娈碉紙bare 缁勫父瑙侊細妯″瀷涓璺祦寮忚锛 + # 娌℃湁 stop/resume 鎵撴柇锛宔nd_of_turn 杩熻繜涓嶆潵 鈫 鏁存涓㈠け锛 + tp = getattr(self, "_turn_printer", None) + if tp is not None and getattr(tp, "_in_turn", False): + pending = "".join(getattr(tp, "_turn_text", [])) + if pending: + end_ms = self.live_rec.session_ms() + start_ms = (self._turn_start_ms + if self._turn_start_ms is not None + else max(end_ms - 3000, 0)) + self.live_rec.log_turn_text(tp.turn_idx, pending, + bool(tp._turn_is_listen)) + self.live_rec.log_subtitle( + start_ms=start_ms, end_ms=end_ms, text=pending, + is_listen=bool(tp._turn_is_listen), turn_idx=tp.turn_idx) + LOG.info("[LIVE] 鍐插嚭鏈敹灏剧殑 turn #%d (%d瀛)", + tp.turn_idx, len(pending)) + except Exception as e: + LOG.warning("[LIVE] flush pending turn err: %s", e) + try: + self.live_rec.stop() + LOG.info("[LIVE] -o record stopped") + except Exception as e: + LOG.warning("[LIVE] stop err: %s", e) + await self._tcp_img._close() + if self._probe is not None: + self._probe.close() + if self._recorder is not None: + self._recorder.close() + await super().close() + + +def parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser(description="ESP32 Phase B Harness runtime") + # 鈹鈹 ESP32 璁惧 鈹鈹 + p.add_argument("--esp32-host", required=True, help="ESP32 IP address") + p.add_argument("--rotate", type=int, default=0, choices=(0, 90, 180, 270), + help="鎽勫儚澶撮『鏃堕拡鏃嬭浆瑙掑害锛屾寜鐪奸暅渚ц鏂瑰悜濉" + "锛堝彇鑷 devices.json 鐨 rotate锛夈傚湪杩涙紡鏂楀墠杞锛" + "鍚﹀垯鏂瑰悜妫娴嬩細鎶婄浉鏈轰晶瑁呰鍒ゆ垚鐢ㄦ埛鎷垮弽浜嗐") + p.add_argument("--esp32-port", type=int, default=80, help="ESP32 HTTP/WS port") + p.add_argument("--image-tcp-port", type=int, default=5000, help="ESP32 TCP image port") + p.add_argument("--image-interval-s", type=float, default=1.0, help="鍙栧浘闂撮殧(绉)") + p.add_argument("--image-timeout-s", type=float, default=1.0, help="鍗曟鍙栧浘瓒呮椂(绉)") + # 鈹鈹 gateway / harness锛堜笌 rokid 涓鑷达級鈹鈹 + p.add_argument("--gateway", default="localhost:8040") + p.add_argument("--gateway-proto", default="auto", + choices=("auto", "duplex", "realtime"), + help="gateway 鍗忚銆俤uplex=/ws/duplex锛圴1锛夛紱" + "realtime=/v1/realtime锛圴2锛孷2 gateway 鍙湁杩欎釜绔偣锛" + "杩 /ws/duplex 浼氳 403 鎷掔粷锛夛紱" + "auto=鎸夌鍙g寽锛8006鈫抮ealtime锛屽叾浣欌啋duplex锛") + p.add_argument("--gateway-tls", action="store_true", default=True) + p.add_argument("--no-tls", dest="gateway_tls", action="store_false") + p.add_argument("--harness-url", default="ws://127.0.0.1:8021/ws/control") + p.add_argument("--client-id", default="esp32-phase-b") + p.add_argument("--skills-config", default=default_skills_config()) + p.add_argument("--cleanup-mode", choices=("light", "full"), default="light") + p.add_argument("--chunk-ms", type=int, default=1_000) + p.add_argument("--force-listen-count", type=int, default=3) + p.add_argument("--audio-queue-packets", type=int, default=96) + p.add_argument("--input-gain", type=float, default=12.0) + p.add_argument("--prompt", default=None, help="system prompt锛堝彲閫夛級") + p.add_argument("--no-play", action="store_true") + p.add_argument("--log-level", default="INFO") + # 鈹鈹 鏃跺簭鎺㈤拡锛堢函鏃佽矾锛夆攢鈹 + p.add_argument("--timing-probe", action="store_true", help="寮鍚椂搴忔帰閽堬紝鍐 CSV") + p.add_argument("--timing-csv", default=None, help="鎺㈤拡 CSV 璺緞锛堥粯璁よ嚜鍔ㄥ甫鏃堕棿鎴筹級") + # 鈹鈹 CV 婕忔枟 鈹鈹 + p.add_argument("--funnel", action="store_true", help="寮鍚 CV 婕忔枟锛堟姄N甯у垽瀹氾紝鍚堟牸鎵嶉佹ā鍨嬶級") + p.add_argument("--scene", default="medicine", choices=("medicine", "stationery"), + help="婕忔枟鍦烘櫙鍙傛暟") + p.add_argument("--funnel-frames", type=int, default=3, help="姣忚疆鎶撳抚鏁帮紙3閫2锛") + p.add_argument("--no-focus", dest="funnel_focus", action="store_false", default=True, + help="鍏抽棴婕忔枟鍐呯殑鑷姩瀵圭劍瑙﹀彂") + # 鈹鈹 session 褰曞埗锛堝榻 -v 鏍煎紡锛夆攢鈹 + p.add_argument("--record", action="store_true", help="褰 session锛堟暣绨囧抚+鍒ゅ畾鍒 sessions/锛") + p.add_argument("--sessions-root", default="sessions", help="session 鏍圭洰褰") + p.add_argument("--no-reject", action="store_true", + help="娑堣瀺锛氭紡鏂楀彧閫 best銆佹案杩滄斁琛岋紝涓嶆嫆缁濅笉鎾姤" + "锛堢敤浜庢妸宸ヤ綔鈶¢夊浘鐨勬敹鐩婁笌宸ヤ綔鈶㈡嫆缁濆垎寮锛") + p.add_argument("--rerun-tail-wait-s", type=float, default=30.0, + help="rerun 鍥炬斁瀹屽悗锛屾渶澶氬啀绛夋ā鍨嬭瀹岀殑绉掓暟锛堥粯璁30锛") + p.add_argument("--record-no-media", action="store_true", + help="鎵归噺璇勫垎鐢細鍙啓 wav+jsonl+transcript锛岃烦杩 jpg 钀界洏鍜 mp4 鎷兼帴") + p.add_argument("--record-live", action="store_true", + help="-o record锛氬綍闊+鍥(鍙瓨鐪熷彂閫佺殑best)鍒 live_sessions/锛屼緵 -o rerun") + p.add_argument("--rerun-from", default=None, + help="-o rerun锛氫粠 live_sessions/ 鍥炴斁(闊+best鍥)閲嶈窇 -o 妯″瀷") + p.add_argument("--funnel-rerun-from", default=None, + help="funnel rerun锛氫粠 live_sessions/ 鍥炴斁(闊+鏁寸皣澶氬抚)閲嶈窇婕忔枟+鎾姤+-o") + p.add_argument("--web-ui-port", type=int, default=None, + help="寮 bridge_ui 绗竴瑙嗚(濡 8080)锛宭ive/rerun/funnel-rerun 閮藉彲鐪") + # 鈹鈹 寮哄埗鎺柦锛堥槻骞昏锛夛細reject 鏃剁敤妯″瀷闊宠壊蹇垫彁绀 + 鍋/鎭㈠璧 8021 鈹鈹 + p.add_argument("--web-ui-host", default="127.0.0.1", + help="绗竴瑙嗚鏈嶅姟鐨勭粦瀹氬湴鍧銆傞粯璁ゅ彧缁戝洖鐜 鈥斺 璇ユ湇鍔℃棤璁よ瘉鍦版彁渚" + "鐢婚潰銆乻ession 鍏冩暟鎹拰鍘熷 user/AI 褰曢煶锛" + "瑕佺粰灞鍩熺綉鍐呭埆鐨勮澶囩湅鎵嶅~ 0.0.0.0锛岄闄╄嚜璐") + p.add_argument("--force-measure", action="store_true", + help="寮鍚己鍒舵帾鏂斤細reject鈫掑仠duplex鈫掓挱棰勭敓鎴恮av鈫掓仮澶峝uplex") + p.add_argument("--reject-wav-dir", default="assets/reject_wav", + help="棰勭敓鎴愮殑 reject 鎻愮ず wav 鐩綍锛坓en_reject_wavs.py 鐢熸垚锛") + return p.parse_args() + + +def main() -> None: + args = parse_args() + logging.basicConfig( + level=getattr(logging, args.log_level.upper(), logging.INFO), + format="%(asctime)s %(levelname)s %(name)s: %(message)s", + force=True, + ) + config = RokidRuntimeConfig( + gateway=args.gateway, + gateway_tls=args.gateway_tls, + harness_url=args.harness_url, + harness_client_id=args.client_id, + skills_config=args.skills_config, + cleanup_mode=args.cleanup_mode, + chunk_ms=args.chunk_ms, + force_listen_count=args.force_listen_count, + audio_queue_packets=args.audio_queue_packets, + input_gain=args.input_gain, + play_audio=not args.no_play, + ) + probe = TimingProbe( + enabled=args.timing_probe, + path=args.timing_csv, + chunk_ms=args.chunk_ms, + ) + funnel = None + if args.funnel: + # rerun 鏃跺己鍒跺叧瀵圭劍锛歳erun 娌℃湁鐪熷疄鐩告満锛宼rigger_af 瀵瑰綍鍒剁殑鍥炬棤鎰忎箟锛 + # 涓斿鐒﹂噸鎶撲細璁 run_once 澶氭姄涓绨 鈫 RecordedImageClient 澶氭秷鑰椾竴涓 round + # 鈫 鍥炬彁鍓嶇敤鍏夛紙鏈夋紡鏂楁瘮鏃犳紡鏂楀厛鐢ㄥ厜鐨勬牴鍥狅級銆 + _is_rerun = bool(args.rerun_from or args.funnel_rerun_from) + _focus = args.funnel_focus and not _is_rerun + funnel = FunnelGate( + args.esp32_host, + scene=args.scene, + n_frames=args.funnel_frames, + enable_focus=_focus, + ) + LOG.info("CV 婕忔枟宸插紑鍚: scene=%s frames=%d focus=%s%s", + args.scene, args.funnel_frames, _focus, + "锛坮erun 寮哄埗鍏冲鐒︼級" if (_is_rerun and args.funnel_focus) else "") + recorder = None # 鏃 SessionRecorder 宸插簾寮冿紝缁熶竴璧 LiveRecorder + # 缁熶竴 record锛堥煶+鏁寸皣澶氬抚+best鏍囪锛夛細--record 鎴 --record-live 閮借Е鍙戙 + live_record_dir = None + if args.record or args.record_live or args.record_no_media: + import time as _t + live_record_dir = os.path.join("live_sessions", _t.strftime("%Y%m%d_%H%M%S")) + # 鏃ュ織鍚屾椂钀借繘 session 鐩綍锛氳瘎鍒嗘椂瑕佹寜鏃堕棿杞村榻 reject/鎾姤/妯″瀷杈撳嚭锛 + # 鎺у埗鍙版棩蹇楁槸婊氬姩鐨勩佸拰 session 鍒嗙锛岃窇澶氱粍 rerun 鏋佹槗瀵归敊銆 + try: + os.makedirs(live_record_dir, exist_ok=True) + _fh = logging.FileHandler( + os.path.join(live_record_dir, "run.log"), encoding="utf-8") + _fh.setFormatter(logging.Formatter( + "%(asctime)s %(levelname)s %(name)s: %(message)s")) + logging.getLogger().addHandler(_fh) + LOG.info("[LIVE] 鏃ュ織鍚屾椂鍐欏叆 %s", os.path.join(live_record_dir, "run.log")) + except Exception as e: + LOG.warning("[LIVE] run.log 鍒涘缓澶辫触: %s", e) + os.makedirs(live_record_dir, exist_ok=True) + LOG.info("缁熶竴 record 宸插紑鍚: %s锛堥煶+鏁寸皣澶氬抚+best鏍囪锛屼緵涓夌 rerun锛", live_record_dir) + # 瑙f瀽 gateway "host:port"锛堝康鎻愮ず鐨 chat TTS 璧板悓涓涓 gateway 鐨 wss锛 + _gw = args.gateway.split("://")[-1] + _gw_host, _, _gw_port = _gw.partition(":") + _gw_host = _gw_host or "127.0.0.1" + _gw_port = int(_gw_port or "8040") + # 閫 gateway 鍗忚銆俈2 gateway 鍙湁 /v1/realtime锛堣繛 /ws/duplex 浼 403锛夛紝 + # V1 鍙湁 /ws/duplex銆俛uto 鎸夌鍙g寽锛8006 鏄 V2 鐨勯粯璁ょ鍙c + _proto = args.gateway_proto + if _proto == "auto": + _proto = "realtime" if _gw_port == 8006 else "duplex" + _factory = None + if _proto == "realtime": + from .realtime_session import RealtimeDuplexSession as _factory + LOG.info("[GW] 鍗忚: /v1/realtime (V2)") + else: + LOG.info("[GW] 鍗忚: /ws/duplex (V1)") + + runtime = PhaseBEsp32Runtime( + config, + **({"session_factory": _factory} if _factory else {}), + esp32_host=args.esp32_host, + esp32_port=args.esp32_port, + image_tcp_port=args.image_tcp_port, + image_interval_s=args.image_interval_s, + image_timeout_s=args.image_timeout_s, + probe=probe, + funnel=funnel, + recorder=recorder, + force_measure=args.force_measure, + reject_wav_dir=args.reject_wav_dir, + gateway_host=_gw_host, + gateway_port=_gw_port, + live_record_dir=live_record_dir, + rerun_from=args.rerun_from, + funnel_rerun_from=args.funnel_rerun_from, + web_ui_port=args.web_ui_port, + web_ui_host=args.web_ui_host, + record_no_media=args.record_no_media, + rerun_tail_wait_s=args.rerun_tail_wait_s, + no_reject=args.no_reject, + rotate=args.rotate, + ) + if args.force_measure: + LOG.info("寮哄埗鎺柦宸插紑鍚: reject鈫抐unnel.stop+蹇垫彁绀, good鈫抐unnel.resume " + "(鑺傛祦5s, chat TTS via wss://%s:%d)", _gw_host, _gw_port) + + LOG.info("ESP32 Phase B input: ws://%s:%d/ws_audio_v2 + TCP:%d", + args.esp32_host, args.esp32_port, args.image_tcp_port) + LOG.info("Harness: %s", config.harness_url) + LOG.info("Gateway: %s://%s", "wss" if config.gateway_tls else "ws", config.gateway) + if args.rotate: + LOG.info("[ROTATE] 鎽勫儚澶撮『鏃堕拡 %d掳 杞锛堣繘婕忔枟鍓嶏級", args.rotate) + + async def _run() -> None: + await runtime.start() + stop = asyncio.Event() + + def _sig(*_a): + stop.set() + try: + loop = asyncio.get_running_loop() + for s in (signal.SIGINT, getattr(signal, "SIGBREAK", signal.SIGINT)): + try: + loop.add_signal_handler(s, stop.set) + except (NotImplementedError, ValueError): + signal.signal(s, _sig) + except Exception: + signal.signal(signal.SIGINT, _sig) + # 绛"淇″彿(Ctrl+C)"鎴"runtime 鍐呴儴 stop_evt(rerun 鍥炬斁瀹岃嚜鍔ㄧ粨鏉)"浠讳竴瑙﹀彂 + _internal = getattr(runtime, "_stop_evt", None) + waiters = [asyncio.create_task(stop.wait())] + if _internal is not None: + waiters.append(asyncio.create_task(_internal.wait())) + await asyncio.wait(waiters, return_when=asyncio.FIRST_COMPLETED) + for w in waiters: + w.cancel() + await runtime.close() + + try: + asyncio.run(_run()) + except KeyboardInterrupt: + LOG.info("Interrupted; shutting down") + + +if __name__ == "__main__": + main() diff --git a/extensions/assistive_harness/phase_b/funnel_gate.py b/extensions/assistive_harness/phase_b/funnel_gate.py new file mode 100644 index 0000000..fe683a7 --- /dev/null +++ b/extensions/assistive_harness/phase_b/funnel_gate.py @@ -0,0 +1,218 @@ +"""婕忔枟闂搁棬锛堟帴鍏 -o 閾捐矾锛夈 + +鎶 cam_pipeline_v2 鐨 run_funnel / process_orientation / CamControl 灏佽鎴 +"涓杞紡鏂楀垽瀹"锛屼緵 esp32_runtime 鍦ㄥ浘鍍忓惊鐜噷璋冪敤锛 + + gate = FunnelGate(esp32_host, scene="medicine", n_frames=3) + result = await gate.run_once(capture_fn) # capture_fn: async ()->Optional[bytes] + +杩斿洖 FunnelDecision锛 + send : bool 鏄惁鏀捐缁欐ā鍨 + best : bytes|None 鏀捐鐨勬渶浣冲抚锛堝凡鍋氬掔疆绾犳锛 + reason : str 鍐崇瓥鍘熷洜锛坅ccepted/severe_shake/unstable/need_focus/orient...锛 + hint : str|None 鎷掔粷鏃剁粰鐢ㄦ埛鐨勬彁绀烘枃鏈紙鍏堣緭鍑 text锛孴TS 閫氶亾寰呭畾锛 + timings: dict grab_ms / judge_ms / orient_ms / af_ms / n_frames + +璁捐渚濇嵁锛堥槻婕傜Щ鏂囨。 + cam_pipeline_v2锛夛細 + - 鍒ゅ畾鍞竴 = run_funnel 鐨 frame_score 涓夊嚭鍙 + argmin 褰掑洜 + - need_focus 鈫 瑙﹀彂鍗曟 AF(0x3022) 鈫 绛 AF_SETTLE_MS 鈫 閲嶆姄涓绨 鈫 閲嶅垽 + - 鍊掔疆 = process_orientation锛宖lipped 鏃剁籂姝f垨鎻愮ず + - 鎷掔粷 reason 鈫 HINTS 鏂囨湰 +鏈ā鍧楀彧鍋"杈撳叆绔妸鍏+寮曞"锛屼笉纰版ā鍨嬭緭鍑轰晶锛堥偅鏄 harness 鐨勪簨锛夈 +""" +from __future__ import annotations + +import asyncio +import time +from typing import Awaitable, Callable, Optional + +from . import cam_pipeline_v2 as cp + +AF_SETTLE_MS = 120 # OV5640 鍗曟瀵圭劍 settle锛堜笌 v5 涓鑷达級 + +# reject_reason -> 鐢ㄦ埛鎻愮ず锛堝厛杈撳嚭 text锛汿TS 閫氶亾纭畾鍚庢帴鍚屼竴浠芥枃鏈級 +HINTS = { + "no_frames": "娌℃湁鎷垮埌鐢婚潰锛岃绋嶇瓑", + "severe_shake": "鐢婚潰鏅冨緱鍘夊锛岃鍏堜繚鎸佷笉鍔", + "unstable": "鐢婚潰杩樺湪鏅冨姩锛岃淇濇寔涓嶅姩涓浼氬効", + "too_dark": "鍏夌嚎澶殫锛岃鍒颁寒涓鐐圭殑鍦版柟", + "aimed_wrong": "濂藉儚娌″鍑嗭紝璇峰鍑嗙洰鏍", + "need_focus": "瀵圭劍涓紝璇锋嬁绋充竴涓", + "orient": "鐢婚潰濂藉儚鍙嶄簡锛岃鍊掕繃鏉", +} + + +class FunnelDecision: + __slots__ = ("send", "best", "reason", "hint", "timings", + "frames", "best_index", "af_triggered", "af_ok") + + def __init__(self, send, best, reason, hint, timings, + frames=None, best_index=-1, af_triggered=False, af_ok=None): + self.send = send + self.best = best + self.reason = reason + self.hint = hint + self.timings = timings + self.frames = frames or [] # 鏁寸皣甯э紙瀛 session 鐢級 + self.best_index = best_index # best 鍦ㄧ皣閲岀殑涓嬫爣 + self.af_triggered = af_triggered # 鏈疆鏈夋病鏈夎Е鍙 AF + self.af_ok = af_ok # AF /reg 璇锋眰鏄惁杩斿洖鎴愬姛锛圢one=娌¤Е鍙戯級 + + +class FunnelGate: + def __init__(self, esp32_host: str, scene: str = "medicine", + n_frames: int = 3, frame_gap_s: float = 0.0, + enable_focus: bool = True): + self.scene = scene + self.n_frames = n_frames + self.frame_gap_s = frame_gap_s + self.enable_focus = enable_focus + self._cfg = (cp.make_scene_config(scene) + if hasattr(cp, "make_scene_config") else cp.FunnelConfig()) + # 鐩告満 HTTP 鎺у埗璧 /control銆/reg 鍒 ESP32锛堝悓 esp32_host锛夈 + # camctl 鎬绘槸鏋勯狅細鍒嗚鲸鐜(set_resolution)鐙珛浜庡鐒︼紝鍗充娇涓嶅鐒︿篃瑕佽 HD銆 + # 瀵圭劍(trigger_af)鍙﹀彈 enable_focus 鎺у埗銆 + self._camctl = cp.CamControl(esp32_host) + self._last_af_mono = 0.0 # 涓婃鐪熸瑙﹀彂瀵圭劍鐨勬椂鍒伙紙鍐峰嵈鐢級 + self._af_cooldown_s = 2.0 # 瀵圭劍鍐峰嵈锛2s 鍐呮渶澶氳Е鍙 1 娆★紙閬垮厤姣忕瀵圭劍锛 + + def warmup_orient(self) -> float: + """鏂瑰悜鍒嗙被鍣ㄩ鐑紙鍚屾锛岃皟鐢ㄦ柟鏀剧嚎绋嬮噷锛夈 + + 鍘熻璁★紙-v pc_vlm_v5_funnel锛夛細棣栨 process_orientation 鍚ā鍨嬪姞杞 + oneDNN + 缂栬瘧寮閿锛堝疄娴 ~2s锛屾湰娆℃棩蹇楅噷鐢氳嚦 5.76s锛夛紝鍚姩鏃跺厛璺戜竴寮犲亣鍥撅紝鎶婅繖绗斾竴娆℃ + 寮閿鎸埌鍚姩闃舵锛岃繍琛屾椂 orient_ms 灏辨槸绾帹鐞嗭紙md 璁板綍锛2143ms 鈫 9ms锛夈 + -o 杩佺Щ鏃舵紡浜嗚繖姝 鈫 orient 鐩村埌绗竴娆 accept 鎵嶇幇鍦哄姞杞斤紝鍗′綇閭d竴杞紝 + 涓斿湪姝や箣鍓嶉摼璺嚭涓嶄簡 send銆傝繑鍥為娆¤楁椂(ms)銆 + """ + import numpy as _np + import cv2 as _cv2 + _warm = _np.full((480, 640, 3), 255, dtype=_np.uint8) + ok, buf = _cv2.imencode(".jpg", _warm) + if not ok: + return -1.0 + t0 = time.monotonic() + cp.process_orientation(buf.tobytes()) + return (time.monotonic() - t0) * 1000.0 + + def set_resolution_hd(self, retries: int = 5, gap_s: float = 0.6) -> bool: + """璁 1280脳720锛圲XGA妗o紝杩欏潡 OV5640 鏋氫妇闈炴爣鍑嗭紝HD(11)鏃犳晥銆乁XGA(13)鎵 720p锛夈 + live 鍚姩鏃 ESP32 鍙兘鍒氬氨缁紝/control 鍋跺彂澶辫触 鈫 閲嶈瘯鍑犳锛岄伩鍏嶈鎵嬪姩杩涚綉椤佃銆 + 澶辫触鏃舵墦璇︾粏鍘熷洜锛坰tatus/寮傚父锛夛紝涓嶅啀闈欓粯銆""" + import logging as _lg + _log = _lg.getLogger("funnel_gate") + for i in range(max(1, retries)): + try: + ok = self._camctl.set_resolution("UXGA") + if ok: + if i > 0: + _log.info("[funnel_gate] set_resolution(UXGA) 绗%d娆¢噸璇曟垚鍔", i + 1) + return True + _log.warning("[funnel_gate] set_resolution(UXGA) 杩斿洖闈200 (绗%d/%d娆)", + i + 1, retries) + except Exception as e: + _log.warning("[funnel_gate] set_resolution(UXGA) 寮傚父 (绗%d/%d娆): %r", + i + 1, retries, e) + time.sleep(gap_s) + return False + + async def _grab_burst( + self, capture_fn: Callable[[], Awaitable[Optional[bytes]]], n: int + ) -> list[bytes]: + frames: list[bytes] = [] + for i in range(n): + jpeg = await capture_fn() + if jpeg: + frames.append(jpeg) + if self.frame_gap_s > 0 and i < n - 1: + await asyncio.sleep(self.frame_gap_s) + return frames + + async def run_once( + self, capture_fn: Callable[[], Awaitable[Optional[bytes]]] + ) -> FunnelDecision: + """璺戜竴杞紡鏂楋細鎶 N 甯 -> run_funnel -> (need_focus 鍒欏鐒﹂噸鎶) -> 鍊掔疆 -> 鍐崇瓥銆""" + timings: dict = {} + + # 1) 鎶撲竴绨 + t0 = time.monotonic() + frames = await self._grab_burst(capture_fn, self.n_frames) + timings["grab_ms"] = round((time.monotonic() - t0) * 1000, 1) + timings["n_frames"] = len(frames) + if not frames: + return FunnelDecision(False, None, "no_frames", HINTS["no_frames"], timings) + + # 2) run_funnel 鍒ゅ畾锛坮un_funnel 鏄悓姝ョ殑锛屾斁绾跨▼姹犻伩鍏嶉樆濉炰簨浠跺惊鐜級 + tj = time.monotonic() + res = await asyncio.to_thread(cp.run_funnel, frames, self._cfg) + timings["judge_ms"] = round((time.monotonic() - tj) * 1000, 1) + timings["af_ms"] = 0.0 + af_triggered = False + af_ok = None + + # 3) need_focus -> 瑙﹀彂鍗曟 AF -> 绛 settle -> 閲嶆姄 -> 閲嶅垽 + # 瀵圭劍鍐峰嵈锛歯eed_focus 浣嗚窛涓婃瀵圭劍 < _af_cooldown_s(2s) 鏃朵笉閲嶅瑙﹀彂锛 + # 缁欏鐒︽椂闂寸敓鏁堬紙姣忕瀵圭劍澶揩锛岄┈杈惧弽澶嶅姩鍙嶈屽涓嶅ソ锛夈傚喎鍗村唴浠 need_focus + # 灏辩敤褰撳墠甯у垽瀹氾紙璇 reject 灏 reject锛屼笉閲嶆姄锛夈 + if res.need_focus and self.enable_focus and self._camctl is not None: + _now = time.monotonic() + if _now - self._last_af_mono >= self._af_cooldown_s: + self._last_af_mono = _now + taf = _now + af_triggered = True + af_ok = await asyncio.to_thread(self._camctl.trigger_af) # True/False + await asyncio.sleep(AF_SETTLE_MS / 1000.0) + frames2 = await self._grab_burst(capture_fn, self.n_frames) + if frames2: + res = await asyncio.to_thread(cp.run_funnel, frames2, self._cfg) + frames = frames2 + timings["af_ms"] = round((time.monotonic() - taf) * 1000, 1) + timings["n_frames"] = len(frames) + else: + # 鍐峰嵈鏈熷唴锛氫笉閲嶅瀵圭劍锛岀敤褰撳墠鍒ゅ畾缁撴灉 + timings["af_cooldown"] = round(self._af_cooldown_s - (_now - self._last_af_mono), 2) + timings["af_ok"] = af_ok + + def _mk(send, best, reason, hint): + bi = frames.index(best) if (best in frames) else -1 + return FunnelDecision(send, best, reason, hint, timings, + frames=frames, best_index=bi, + af_triggered=af_triggered, af_ok=af_ok) + + # 4) 鎷掔粷鍑哄彛 + if not res.accepted: + reason = res.reject_reason or "reject" + # cam_pipeline 鐨勬嫆缁濆嚭鍙e彧璁 best_index锛堟敞閲婏細浠呬緵鍙傝, 涓嶅杺妯″瀷锛夛紝 + # 涓嶈 best_jpg 鈥斺 杩欐槸鍘熻璁℃剰鍥撅細鎷掔粷灏变笉閫佸浘銆 + # 浣嗘秷铻 arm锛--no-reject锛氬彧閫 best銆佹案杩滄斁琛岋級闇瑕佹嬁鍒拌繖涓甯э紝 + # 鎵浠ヨ繖閲屾寜 best_index 鎶婂抚琛ュ嚭鏉ャ**send 浠嶇劧鏄 False**锛 + # 姝e父閾捐矾瀹屽叏涓嶅彈褰卞搷锛氬彧鏈夋樉寮忓紑浜 --no-reject 鎵嶄細鍘荤敤瀹冦 + _bj = res.best_jpg + if _bj is None: + bi = getattr(res, "best_index", -1) + if isinstance(bi, int) and 0 <= bi < len(frames): + _bj = frames[bi] + return _mk(False, _bj, reason, + HINTS.get(reason, "鐪嬩笉娓呮锛岃璋冩暣涓涓")) + + # 5) 鎺ュ彈甯х殑鏂瑰悜妫娴嬶紙OCR 鏂瑰悜鍒嗙被鍣ㄤ紭鍏堬紝瑙 cam_pipeline.process_orientation锛 + # 鍘熷垯銆屽畞鎷掔粷涓嶅康閿欍嶏細鏂瑰悜涓嶆 -> 鎷掔粷 + 鎻愮ず鐢ㄦ埛杞锛岀粷涓嶈嚜鍔ㄧ籂姝c佺粷涓嶉佸掑浘銆 + # process_orientation 鍙瘖鏂柟鍚戙佽繑鍥炵殑 jpg 鏈棆杞紝鏁呬笉鑳芥嬁瀹冨綋"绾犳鍚"閫佹ā鍨嬨 + best = res.best_jpg + to_ = time.monotonic() + try: + _jpg, geom = await asyncio.to_thread(cp.process_orientation, best) + except Exception: + geom = {"ok": False} + timings["orient_ms"] = round((time.monotonic() - to_) * 1000, 1) + + orient_state = geom.get("orient_state") if isinstance(geom, dict) else None + orient_hint = geom.get("orient_hint") if isinstance(geom, dict) else None + + # upright 鎵嶆斁琛岋紱flipped/sideways 鎷掔粷骞舵彁绀猴紱uncertain 涔熸斁琛岋紙涓嶇‖鎷︼紝閬垮厤璇嫆锛 + if orient_state in ("flipped", "sideways"): + hint = orient_hint or HINTS["orient"] + return _mk(False, best, "orient_" + orient_state, hint) + + # upright / uncertain / 妫娴嬩笉鍙敤 -> 鏀捐鍘熷浘锛堜笉鍋氫换浣曟棆杞級 + return _mk(True, best, "send", None) diff --git a/extensions/assistive_harness/phase_b/realtime_session.py b/extensions/assistive_harness/phase_b/realtime_session.py new file mode 100644 index 0000000..c0e4435 --- /dev/null +++ b/extensions/assistive_harness/phase_b/realtime_session.py @@ -0,0 +1,306 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +""" +realtime_session.py 鈥斺 鎶 duplex 浼氳瘽鎺ュ埌 V2 gateway 鐨 /v1/realtime 鍗忚銆 + +涓轰粈涔堥渶瑕佸畠 +------------ +V1 gateway 鏈 `/ws/duplex/{session_id}`锛**V2 gateway 鍙湁 `/v1/realtime`** +锛堝疄娴嬶細V2 gateway.py 閲屽敮涓鐨 @app.websocket 鏄 /v1/realtime锛 + 杩 /ws/duplex/... 浼氳 FastAPI 鐩存帴 403 鎷掔粷鎻℃墜锛夈 +鎵浠 rokid_runtime / esp32_runtime 閲岄偅濂 GatewayDuplexSession 鍦 V2 涓嬭繛涓嶄笂锛 +demo_esp32_duplex_0703 鍚屾牱杩炰笉涓婏紙瀹冭繛鐨勪篃鏄 /ws/duplex锛夈 + +鍋氭硶 +---- +`GatewaySessionManager` 鏈夌幇鎴愮殑鏇挎崲鐐 `session_factory`锛 +鎵浠ヨ繖閲**缁ф壙 GatewayDuplexSession锛屽彧瑕嗙洊鍗忚鐩稿叧鐨勬柟娉**锛 +鍏朵綑锛堥煶棰戞敀鍧椼佺姸鎬佺鐞嗐乻tart/_run 楠ㄦ灦銆乻top 鐨勬敹灏撅級鍏ㄩ儴澶嶇敤銆 +rokid_runtime.py 鍜 esp32_runtime.py 閮戒笉鐢ㄦ敼锛屽彧鍦ㄦ瀯閫 manager 鏃朵紶鍏ユ湰绫汇 + +鍗忚瀵圭収锛圴1 鍐呴儴 鈫 V2 瀵瑰锛 +------------------------------ + 杩炴帴 URL 甯 session_id 鈫 杩炰笂鍚 session.init锛屾湇鍔$鍥 session.created + 鎺掗槦 queued / queue_done 鈫 session.queued / session.queue_done + 涓婅 audio_chunk 鈫 input.append + audio_base64 鈫 input.audio + frame_base64_list 鈫 input.video_frames + force_listen 鈫 input.force_listen + 涓嬭 {is_listen,text,audio_data}鈫 response.output.delta 鐨 kind: listen/text/audio + 缁撴潫 stop 鈫 session.close + +娉ㄦ剰 +---- +路 V2 鐨 session_id 鐢辨湇鍔$鍦 session.created 閲岀粰锛屽鎴风涓嶈兘鑷繁鎸囧畾銆 + 浣 manager/鏃ュ織閲屽埌澶勭敤 self.session_id锛屾墍浠ヤ繚鐣欐湰鍦扮敓鎴愮殑閭d釜浣滀负鍗犱綅锛 + 鎷垮埌鏈嶅姟绔殑涔嬪悗瑕嗙洊鎺夈 +路 full-duplex 涓嬫病鏈 response.done锛堟枃妗h瀹冨彧鐢ㄤ簬 mode=chat锛夛紝 + turn 杈圭晫闈 delta 閲岀殑 turn_id 鍙樺寲鍒ゆ柇銆 +路 kind=audio 鐨 PCM 鏄 24kHz mono float32 base64锛屽拰 V1 鐨 audio_data 鍚屾牸寮忥紝 + 鎵浠 on_result 閭d竴渚т笉鐢ㄦ敼銆 +""" +from __future__ import annotations + +import asyncio +import base64 +import json +import time +from typing import Any + +import aiohttp +import numpy as np + +from .rokid_runtime import ( + LOG, + GatewayDuplexSession, + SAMPLE_RATE_IN, + float32_to_base64, + now_ms, +) + + +class RealtimeDuplexSession(GatewayDuplexSession): + """V2 /v1/realtime 鍗忚鐗堢殑 duplex 浼氳瘽銆傛瀯閫犵鍚嶄笌鐖剁被瀹屽叏涓鑷淬""" + + # ------------------------------------------------------------ 杩炴帴 + async def _run(self) -> None: + scheme = "wss" if self.config.gateway_tls else "ws" + # mode=video锛氳繛缁煶棰 + 鍙甫瑙嗛甯э紝300s 涓婇檺銆 + # mode=audio 鏄函闊抽锛600s锛夛紝鎴戜滑瑕侀佸浘鎵浠ョ敤 video銆 + url = f"{scheme}://{self.config.gateway}/v1/realtime?mode=video" + self.status = "connecting" + LOG.info("[GW] connecting generation=%d %s (realtime)", + self.spec.generation, url) + try: + async with aiohttp.ClientSession() as client: + async with client.ws_connect( + url, + heartbeat=30, + ssl=self._ssl_context(), + max_msg_size=0, + ) as ws: + self.ws = ws + await self._prepare(ws) + self.status = "running" + if self._ready is not None and not self._ready.done(): + self._ready.set_result(None) + await asyncio.gather( + self._send_loop(ws), + self._receive_loop(ws), + ) + except asyncio.CancelledError: + raise + except Exception as exc: # noqa: BLE001 + self.status = "failed" + self.last_error = str(exc) + LOG.warning("[GW] session %s failed: %s", self.session_id, exc) + if self._ready is not None and not self._ready.done(): + self._ready.set_exception(exc) + finally: + self._stopped.set() + if self.status not in ("failed",): + self.status = "closed" + + # ------------------------------------------------------------ 鎻℃墜 + async def _prepare(self, ws: aiohttp.ClientWebSocketResponse) -> None: + """绛 session.queue_done 鈫 鍙 session.init 鈫 绛 session.created銆 + + 鏂囨。鏄庣‘瑕佹眰锛**蹇呴』绛 queue_done 鍐嶅彂 session.init**銆 + 娌℃湁鎺掗槦鏃舵湇鍔$浼氱珛鍒讳笅鍙 queue_done銆 + """ + self.status = "queued" + while True: + message = await ws.receive() + if message.type != aiohttp.WSMsgType.TEXT: + if message.type in (aiohttp.WSMsgType.CLOSED, aiohttp.WSMsgType.ERROR): + raise RuntimeError("gateway closed while queued") + continue + payload = json.loads(message.data) + mtype = payload.get("type") + if mtype == "session.queue_done": + break + if mtype == "error": + raise RuntimeError( + (payload.get("error") or {}).get("message") + if isinstance(payload.get("error"), dict) + else payload.get("error") or "gateway queue error" + ) + if mtype in ("session.queued", "session.queue_update"): + LOG.info("[GW] queue position=%s eta=%s", + payload.get("position"), + payload.get("estimated_wait_s")) + + self.status = "preparing" + init_payload: dict[str, Any] = { + "system_prompt": self.spec.system_prompt, + "config": { + "force_listen_count": self.config.force_listen_count, + "chunk_ms": self.config.chunk_ms, + "generate_audio": True, + "max_new_speak_tokens_per_chunk": ( + self.config.max_new_speak_tokens_per_chunk + ), + "length_penalty": self.config.length_penalty, + }, + } + await self._send_json({"type": "session.init", "payload": init_payload}) + + while True: + message = await ws.receive() + if message.type != aiohttp.WSMsgType.TEXT: + if message.type in (aiohttp.WSMsgType.CLOSED, aiohttp.WSMsgType.ERROR): + raise RuntimeError("gateway closed while preparing") + continue + payload = json.loads(message.data) + mtype = payload.get("type") + if mtype == "session.created": + # session_id 鐢辨湇鍔$缁欙紝瑕嗙洊鏈湴鍗犱綅鐨勯偅涓 + sid = payload.get("session_id") + if sid: + self.session_id = sid + LOG.info("[GW] prepared session=%s generation=%d skill=%s mode=%s", + self.session_id, self.spec.generation, + self.spec.skill_id, payload.get("mode")) + return + if mtype == "error": + raise RuntimeError( + (payload.get("error") or {}).get("message") + if isinstance(payload.get("error"), dict) + else payload.get("error") or "gateway prepare error" + ) + + # ------------------------------------------------------------ 涓婅 + async def _send_loop(self, ws: aiohttp.ClientWebSocketResponse) -> None: + """鏀 1s 闊抽 + 甯︿笂鏈鏂板抚锛屽彂 input.append銆 + + 鏀掑潡銆佸彇甯с乫orce_listen 鐨勫垽鏂昏緫涓庣埗绫诲畬鍏ㄤ竴鑷达紝鍙崲娑堟伅澶栧3銆 + """ + carry = bytearray() + last_frame_sequence = -1 + last_frame_sent = 0.0 + sent = 0 + while not self._closing: + audio = await self._next_audio_chunk(carry) + if audio is None: + continue + inp: dict[str, Any] = { + "audio": float32_to_base64(audio), + "max_slice_nums": self.config.max_slice_nums, + } + if self.gate.speech_hold_active: + inp["force_listen"] = True + frame = self.latest_frame + frame_age_ms = now_ms() - frame.timestamp_ms + frame_due = time.monotonic() - last_frame_sent >= self.config.image_resend_s + if ( + frame.jpeg + and frame_age_ms <= self.config.image_max_age_s * 1000 + and (frame.sequence != last_frame_sequence or frame_due) + ): + inp["video_frames"] = [base64.b64encode(frame.jpeg).decode("ascii")] + last_frame_sequence = frame.sequence + last_frame_sent = time.monotonic() + sent += 1 + LOG.debug("[GW-SEND] #%d audio=%.2fs lvl=%.4f frame=%s hold=%s", + sent, len(audio) / SAMPLE_RATE_IN, + float(np.abs(audio).mean()), + "video_frames" in inp, self.gate.speech_hold_active) + await self._send_json({"type": "input.append", "input": inp}) + + # ------------------------------------------------------------ 涓嬭 + async def _receive_loop(self, ws: aiohttp.ClientWebSocketResponse) -> None: + """鎶 response.output.delta 缈昏瘧鍥 {is_listen, text, audio_data}銆 + + 鐖剁被鐨 on_result 鎷垮埌鐨勬槸 V1 閭e瀛楁锛岃繖閲屼繚鎸佷笉鍙橈紝 + 鎵浠 handle_result / TurnPrinter / speaker 閭d竴渚т竴琛岄兘涓嶇敤鏀广 + """ + cur_turn = None + async for message in ws: + if self._closing: + break + if message.type != aiohttp.WSMsgType.TEXT: + if message.type in (aiohttp.WSMsgType.CLOSED, aiohttp.WSMsgType.ERROR): + break + continue + payload = json.loads(message.data) + mtype = payload.get("type") + + if mtype == "response.output.delta": + kind = payload.get("kind") + turn = payload.get("turn_id") + # full-duplex 娌℃湁 response.done锛岀敤 turn_id 鍙樺寲褰撹竟鐣 + end_of_turn = (cur_turn is not None and turn is not None + and turn != cur_turn) + if turn is not None: + cur_turn = turn + result: dict[str, Any] = { + "type": "result", + "is_listen": kind == "listen", + "text": payload.get("text") or "", + "audio_data": payload.get("audio") or "", + "end_of_turn": end_of_turn, + "turn_id": turn, + } + await self.on_result(self, result) + continue + + if mtype == "response.done": + # 瀹炴祴锛歠ull-duplex 涓**涔熶細**鍙 response.done锛堟枃妗h鍙敤浜 chat锛 + # 涓庡疄闄呬笉绗︼級锛岃屼笖瀹冪殑 text 鏄繖涓娈电殑**瀹屾暣鏂囨湰** 鈥斺 + # 鍓嶉潰 delta 宸茬粡鎶婅繖浜涘瓧閫愬潡浼犱笅鍘讳簡锛岃繖閲屽啀浼犱竴娆″氨鏄噸澶 + # 锛堝疄娴嬫瘡娈甸兘琚涓ら亶锛屽彧宸 1~2ms锛夈 + # 鎵浠ヨ繖閲屽彧鏍囪 turn 缁撴潫锛**涓嶅啀浼 text**銆 + await self.on_result(self, { + "type": "result", "is_listen": False, + "text": "", "audio_data": "", + "end_of_turn": True, "turn_id": cur_turn, + }) + continue + + if mtype == "session.closed": + LOG.info("[GW] session.closed reason=%s", payload.get("reason")) + break + + if mtype == "error": + err = payload.get("error") + msg = err.get("message") if isinstance(err, dict) else str(err) + LOG.warning("[GW] error: %s", msg) + self.last_error = msg or "gateway error" + continue + + if mtype == "debug": + # V2 鑷甫鐨勯樁娈佃拷韪紙llm.chunk / tts.chunk / t2w.chunk锛屽甫 ts锛夈 + # 骞虫椂涓嶆墦锛岄渶瑕佹椂 --log-level DEBUG 鎵撳紑鐪嬫湇鍔$鍚勯樁娈佃楁椂銆 + LOG.debug("[GW-DEBUG] %s", json.dumps(payload, ensure_ascii=False)[:200]) + continue + + # ------------------------------------------------------------ 缁撴潫 + async def stop(self, cleanup_mode: str) -> None: + self._closing = True + if self.ws is not None and not self.ws.closed: + try: + await self._send_json({"type": "session.close", + "reason": cleanup_mode or "user_stop"}) + except Exception: + pass + try: + await asyncio.wait_for(self._stopped.wait(), + timeout=self.config.close_timeout_s) + except asyncio.TimeoutError: + pass + if self.ws is not None and not self.ws.closed: + await self.ws.close() + if self._task and not self._task.done(): + self._task.cancel() + await asyncio.gather(self._task, return_exceptions=True) + + # ------------------------------------------------------------ 鏂囨湰娉ㄥ叆 + async def inject_task(self, text: str) -> bool: + """V2 鐨 input.append 娌℃湁 inject_text 瀛楁锛屾殏涓嶆敮鎸併 + + 鐖剁被闈犲畠瀹炵幇"鎶鑳藉垏鎹㈡椂鎶婃寚浠ゆ枃鏈缁欐ā鍨"銆俈2 鍗忚閲屾病鏈夊搴斿瓧娈碉紝 + 纭浼氳鏈嶅姟绔拷鐣ユ垨鎶ラ敊锛屾墍浠ユ槑纭繑鍥 False 璁╄皟鐢ㄦ柟璧板埆鐨勮矾寰勶紝 + 鑰屼笉鏄亣瑁呮垚鍔熴 + """ + LOG.info("[GW] inject_task 鍦 /v1/realtime 鍗忚涓嬫殏涓嶆敮鎸侊紝璺宠繃: %s", text[:30]) + return False diff --git a/extensions/assistive_harness/phase_b/recorder_live.py b/extensions/assistive_harness/phase_b/recorder_live.py new file mode 100644 index 0000000..3e17528 --- /dev/null +++ b/extensions/assistive_harness/phase_b/recorder_live.py @@ -0,0 +1,860 @@ +"""recorder_live.py 鈥 褰曞睆寮忓疄鏃跺綍鍒跺櫒 v5.0 +================================================= + +v5.0 鏀瑰姩:user 杞ㄤ粠"ring buffer 鍒囩墖鎷兼帴"鏀规垚"WS 鍘熷鍖呯洿褰"銆 +鍜岃摑鐗欒虫満褰曢氳瘽璇箟涓鑷:涓㈠寘 = 璇ユ缂哄け,涓嶅啀琛ラ浂鍘诲榻愬閽熴 +鍜 AI 杞ㄥ畬鍏ㄥ绉扳斺擜I 杞ㄥ綍鐨勬槸 PortAudio DAC 瀹為檯鍐欏嚭鐨勬牱鏈 + +璋冪敤椤哄簭: + live_rec = LiveRecorder(session_dir) + live_rec.attach_to_player(speaker) # 蹇呴』鍦 speaker.start() 涔嬪墠 + speaker.start() + live_rec.start() # 姝ゅ悗 user/ai/frame 鎵嶄細琚褰 + + # ESP32 reader 姣忔敹涓鍖呭氨璋: + live_rec.feed_user_raw(pcm_f32) # 16kHz 鍘熷,涓㈠寘灏变涪 + + # 姣忓彂涓涓 chunk 閰嶇殑鍥: + live_rec.on_frame(jpeg, chunk_idx) + + live_rec.stop() + live_rec.finalize_mp4() +""" + +from __future__ import annotations + +import json +import logging +import shutil +import subprocess +import threading +import time +import wave +from pathlib import Path +from typing import Optional + +import numpy as np + +LOGGER = logging.getLogger("recorder_live") + + +class LiveRecorder: + def __init__( + self, + session_dir: Optional[Path], + user_sr: int = 16000, + ai_sr: int = 24000, + no_media: bool = False, + ): + self.enabled = session_dir is not None + self.dir: Optional[Path] = Path(session_dir) if session_dir else None + self.user_sr = user_sr + self.ai_sr = ai_sr + # no_media锛氭壒閲 rerun 璇勫垎鍙渶瑕 wav + jsonl + transcript锛 + # 璺宠繃 jpg 钀界洏鍜 mp4 鎷兼帴锛150 娆 rerun 浼氶噸澶嶅鍒舵暣绨囧浘銆佽窇 450 娆 ffmpeg锛夈 + self.no_media = bool(no_media) + + self._t0: Optional[float] = None + + # User 杞:ESP32 WS 鏉ョ殑鍘熷 PCM,涓㈠寘 = 缂哄け,涓嶈ˉ闆 + self._user_t_start: Optional[float] = None + # self._user_chunks: list[np.ndarray] = [] + self._user_chunks: list[tuple[float, np.ndarray]] = [] + # AI 杞:PortAudio DAC 瀹為檯杈撳嚭(鍚 underrun 鏃跺~鐨勯浂) + self._ai_t_start: Optional[float] = None + #self._ai_chunks: list[np.ndarray] = [] + self._ai_chunks: list[tuple[float, np.ndarray]] = [] + # Frames + self._frames: list[tuple[float, str, int]] = [] + self._funnel_rounds: list[dict] = [] # 姣忚疆鏁寸皣澶氬抚+鍒ゅ畾锛堢粺涓record锛 + self._round_counter: int = -1 # on_funnel_round 杞璁℃暟锛坬id锛屾瘡杞+1锛 + self._transcript = None # transcript.txt 鍙ユ焺 + self._subtitles = None # subtitles.jsonl 鍙ユ焺 + self._chunks_f = None # model_chunks.jsonl 鍙ユ焺 + + self._lock = threading.Lock() + self._started = threading.Event() + self._stopping = threading.Event() + + self._user_calls = 0 + self._ai_calls = 0 + + # ------------------------------------------------------------ + # Lifecycle + # ------------------------------------------------------------ + + def attach_to_player(self, player) -> None: + if not self.enabled: + return + orig_enqueue = player.enqueue + rec = self + + # PCSpeaker.enqueue 鏄 async锛岀鍚 (pcm, generation)銆傚寘瑁呭繀椤诲悓绛惧悕 + async锛 + # 鍚﹀垯妯″瀷闊抽 enqueue 鎶 "takes 1 positional argument but 2 were given"锛 + # 瀵艰嚧 AI 闊虫挱涓嶅嚭銆佷篃褰曚笉杩涳紙live 鍚笉鍒板0闊筹級銆 + async def wrapped_enqueue(pcm_f32, generation=0): + await orig_enqueue(pcm_f32, generation) + if (not rec._started.is_set() or rec._stopping.is_set() + or rec._t0 is None or pcm_f32 is None or pcm_f32.size == 0): + return + try: + pcm = pcm_f32 if pcm_f32.dtype == np.float32 else pcm_f32.astype(np.float32) + pcm = pcm.copy() + t = max(0.0, time.monotonic() - rec._t0) + with rec._lock: + if rec._ai_t_start is None: + rec._ai_t_start = t + rec._ai_chunks.append((t, pcm)) + rec._ai_calls += 1 + except Exception as e: + LOGGER.warning("[LIVE] enqueue hook err: %r", e) + + player.enqueue = wrapped_enqueue + player._orig_enqueue = orig_enqueue # 鏆撮湶鍘熷鏂规硶锛岀粫杩 hook 鐨勫満鏅娇鐢 + LOGGER.info("[LIVE] attached to SPK enqueue (async, pcm+generation)") + + def start(self) -> None: + if not self.enabled or self._started.is_set(): + return + assert self.dir is not None + self.dir.mkdir(parents=True, exist_ok=True) + #(self.dir / "live_images").mkdir(exist_ok=True) + self._t0 = time.monotonic() + self._started.set() + # 璇勫垎鐢細妯″瀷/鐢ㄦ埛鐨勬垚娈垫枃鏈 + 鏃堕棿杞达紙绉绘鑷 demo_esp32_duplex_0703 鐨 + # SessionRecorder.log_turn_text / log_subtitle锛夈 + # transcript.txt 鈥斺 浜鸿锛孾AI #n] / [LISTEN #n] 鏁存 + # subtitles.jsonl 鈥斺 鏈鸿锛寋start_ms,end_ms,text,is_listen,turn_idx} + # is_listen=True 鐨 turn 灏辨槸 8021 ASR 璇嗗埆鍑虹殑鐢ㄦ埛璇磋瘽 鈫 鎻愰棶缁撴潫鏃跺埢 t0 浠庤繖鎷 + try: + self._transcript = (self.dir / "transcript.txt").open("w", encoding="utf-8") + self._transcript.write(f"# session {self.dir.name}\n") + self._transcript.flush() + self._subtitles = (self.dir / "subtitles.jsonl").open("w", encoding="utf-8") + self._chunks_f = (self.dir / "model_chunks.jsonl").open("w", encoding="utf-8") + except Exception as e: + LOGGER.warning("[LIVE] transcript/subtitles 鎵撳紑澶辫触: %s", e) + self._transcript = None + self._subtitles = None + self._chunks_f = None + LOGGER.info( + "[LIVE] recording started: %s (user=%dHz ai=%dHz)", + self.dir, self.user_sr, self.ai_sr, + ) + + def session_ms(self) -> int: + """session 鍐呯浉瀵规绉掞紙涓 frames.jsonl 鐨 t 鍚屼竴鏉℃椂闂磋酱锛夈""" + if self._t0 is None: + return 0 + return int((time.monotonic() - self._t0) * 1000.0) + + def log_model_chunk(self, text: str, is_listen: bool, end_of_turn: bool, + audio_ms: float = 0.0) -> None: + """閫 chunk 鍘熷璁板綍 鈥斺 **涓鏉¢兘涓嶈繃婊**銆 + + 杩欐槸璇婃柇鐢ㄧ殑鍘熷娴佹按锛-o 鏄 1Hz 鍐崇瓥锛宭isten 鏈熼棿澶ч噺 chunk 鏄 + text="" 涓 end_of_turn=False銆備箣鍓嶈繖閲屽啓浜 + if not text and not end_of_turn: return + 鎶婂畠浠叏鎵斾簡锛岀粨鏋 24 绉掔殑浼氳瘽鍙墿 5 鏉★紝鐪嬩笂鍘诲儚"妯″瀷 5.7 绉掓墠鍑轰竴涓 + chunk / 杈撳叆鍒嗗潡鍧忎簡"鈥斺旈偅鏄褰曞嚱鏁伴犲嚭鏉ョ殑鍋囪薄锛屼笉鏄簨瀹炪 + 瑕佸垽鏂ā鍨嬭妭濂忔槸鍚︽甯革紝蹇呴』鐪嬪埌姣忎竴涓 chunk銆 + """ + if not self.enabled or not self._started.is_set() or self.dir is None: + return + if self._chunks_f is None: + return + try: + self._chunks_f.write(json.dumps({ + "session_ms": self.session_ms(), + "text": text, + "is_listen": bool(is_listen), + "end_of_turn": bool(end_of_turn), + "audio_ms": round(float(audio_ms), 1), + }, ensure_ascii=False) + "\n") + self._chunks_f.flush() + except Exception: + pass + + def log_event(self, tag: str, text: str) -> None: + """鎶婃紡鏂椾晶浜嬩欢鍐欒繘 transcript锛屽拰 USER/AI 鍚屼竴鏉℃椂闂磋酱銆 + + 璇勫垎鏃惰鍒"杩欐鎷︽埅瀵瑰簲鍝鍥炵瓟"锛屼笁鑰呭繀椤诲湪鍚屼竴鏉′汉璇绘椂闂磋酱涓婏細 + [USER @12480ms] 杩欎笂闈㈠啓鐨勬槸浠涔 + [REJECT @13520ms] severe_shake 鐢婚潰鏅冨緱鍘夊锛岃鍏堜繚鎸佷笉鍔 + [HINT @13900ms] severe_shake (2.52s) + [AI #3 @16100ms] 涓婇潰鍐欑潃闃胯帿瑗挎灄鑳跺泭... + 鏈鸿鐨勪竴浠戒粛鍦 frames.jsonl锛堟瘡杞 send/reason锛夈 + """ + if not self.enabled or not self._started.is_set() or self._transcript is None: + return + try: + self._transcript.write(f"[{tag} @{self.session_ms()}ms] {text}\n") + self._transcript.flush() + except Exception: + pass + + def log_asr(self, utterance: str, final_at_ms=None, confidence=None) -> None: + """8021 ASR 璇嗗埆鍑虹殑鐢ㄦ埛璇磋瘽銆傝瘎鍒嗗彇 t0 鐢ㄣ + + 鍐欎袱澶勶細transcript.txt 鐨 [USER] 琛岋紙浜鸿锛夈乻ubtitles.jsonl 鐨 + is_listen=true 璁板綍锛堟満璇伙紝涓庢ā鍨 turn 鍚屼竴鏉℃椂闂磋酱銆佸悓涓涓 session_ms 鍩哄噯锛夈 + final_at_ms 鏄 8021 渚х殑鏃堕挓锛屽拰鏈 session 鐨 _t0 涓嶅悓婧愶紝鍙綔鍙傝冨瓨杩 + asr_events.jsonl锛涘榻愪竴寰嬬敤鏈湴 session_ms()銆 + """ + if not self.enabled or not self._started.is_set(): + return + t_ms = self.session_ms() + try: + if self._transcript is not None: + self._transcript.write(f"[USER @{t_ms}ms] {utterance}\n") + self._transcript.flush() + except Exception: + pass + # 澶嶇敤 subtitles 鐨 turn 搴忓垪锛堣礋鏁 turn_idx 鏍囪瘑 ASR锛岄伩鍏嶅拰妯″瀷 turn 鎾炲彿锛 + self._asr_seq = getattr(self, "_asr_seq", 0) + 1 + self.log_subtitle(start_ms=t_ms, end_ms=t_ms, text=utterance, + is_listen=True, turn_idx=-self._asr_seq) + try: + if self.dir is not None: + with (self.dir / "asr_events.jsonl").open("a", encoding="utf-8") as f: + f.write(json.dumps({ + "session_ms": t_ms, + "utterance": utterance, + "harness_final_at_ms": final_at_ms, + "confidence": confidence, + }, ensure_ascii=False) + "\n") + except Exception: + pass + + def log_turn_text(self, turn_idx: int, text: str, is_listen: bool) -> None: + if not self.enabled or self._transcript is None: + return + tag = "LISTEN" if is_listen else "AI" + try: + self._transcript.write(f"[{tag} #{turn_idx} @{self.session_ms()}ms] {text}\n") + self._transcript.flush() + except Exception: + pass + + def log_subtitle(self, start_ms: int, end_ms: int, text: str, + is_listen: bool, turn_idx: int) -> None: + if not self.enabled or self._subtitles is None: + return + try: + self._subtitles.write(json.dumps({ + "start_ms": int(start_ms), + "end_ms": int(end_ms), + "text": text, + "is_listen": bool(is_listen), + "turn_idx": int(turn_idx), + }, ensure_ascii=False) + "\n") + self._subtitles.flush() + except Exception: + pass + + def stop(self) -> None: + if not self.enabled or not self._started.is_set() or self._stopping.is_set(): + return + self._stopping.set() + assert self._t0 is not None and self.dir is not None + total = time.monotonic() - self._t0 + + with self._lock: + user_chunks = list(self._user_chunks) + user_t_start = self._user_t_start + ai_chunks = list(self._ai_chunks) + ai_t_start = self._ai_t_start + n_frames = len(self._frames) + + u_path = self.dir / "live_user.wav" + a_path = self.dir / "live_ai.wav" + u_dur = self._write_track_wav(u_path, user_t_start, user_chunks, + self.user_sr, total) + a_dur = self._write_track_wav(a_path, ai_t_start, ai_chunks, + self.ai_sr, total) + + u_size = u_path.stat().st_size if u_path.exists() else 0 + a_size = a_path.stat().st_size if a_path.exists() else 0 + + # 鍐 events.jsonl锛堜緵 rerun 鐨 LocalImageSource 璇伙細chunk_idx 鈫 img 鏄犲皠锛 + try: + import json as _json + with self._lock: + frames_snap = list(self._frames) + ev_path = self.dir / "events.jsonl" + with ev_path.open("w", encoding="utf-8") as ef: + for (ft, frel, fidx) in frames_snap: + ef.write(_json.dumps( + {"kind": "chunk_sent", "idx": int(fidx), + "img": Path(frel).name, "t": round(ft, 3)}, + ensure_ascii=False) + "\n") + LOGGER.info("[LIVE] events.jsonl written: %d frames", len(frames_snap)) + except Exception as e: + LOGGER.warning("[LIVE] write events.jsonl err: %r", e) + + # 鍐 frames.jsonl锛堟暣绨囧甯 + 鍒ゅ畾锛屼緵 funnel rerun 閲嶅垽 / 鏃犳紡鏂 rerun 鍙栦唬琛ㄥ抚锛 + try: + import json as _json2 + with self._lock: + rounds_snap = list(self._funnel_rounds) + fr_path = self.dir / "frames.jsonl" + with fr_path.open("w", encoding="utf-8") as ff: + for rec in rounds_snap: + ff.write(_json2.dumps(rec, ensure_ascii=False) + "\n") + LOGGER.info("[LIVE] frames.jsonl written: %d rounds", len(rounds_snap)) + except Exception as e: + LOGGER.warning("[LIVE] write frames.jsonl err: %r", e) + + u_nz = self._nonzero_ratio(user_chunks) + a_nz = self._nonzero_ratio(ai_chunks) + + LOGGER.info( + "[LIVE] stopped: wall=%.2fs " + "user(calls=%d audio=%.2fs t_start=%.2fs nonzero=%.1f%%) " + "ai(calls=%d audio=%.2fs t_start=%.2fs nonzero=%.1f%%) " + "frames=%d user_wav=%dB ai_wav=%dB", + total, + self._user_calls, u_dur, + user_t_start if user_t_start is not None else -1, u_nz * 100, + self._ai_calls, a_dur, + ai_t_start if ai_t_start is not None else -1, a_nz * 100, + n_frames, u_size, a_size, + ) + + if self._user_calls == 0: + LOGGER.warning( + "[LIVE] feed_user_raw() was NEVER called 鈥 " + "esp32_audio_reader 娌℃妸 live_rec 鎺ヨ繘鏉ャ" + ) + if self._ai_calls == 0: + LOGGER.warning( + "[LIVE] AI callback NEVER fired 鈥 attach_to_player 娌¤涓娿" + "speaker 鏈惎鍔ㄣ佹垨寮浜 --no-play銆" + ) + + # 鍏抽棴璇勫垎鐢ㄦ枃鏈惤鐩 + for _fh in (self._transcript, self._subtitles, self._chunks_f): + try: + if _fh is not None: + _fh.close() + except Exception: + pass + self._transcript = None + self._subtitles = None + self._chunks_f = None + + # 缁撴潫鑷姩钀界洏 mp4锛堜笉鐢ㄦ墜鍔級锛 + # 鈶 鏁寸皣澶氬抚 + 瑙掕惤 reject 鏍囨敞锛堟紨绀轰富鐢紝娴佺晠 + 鍙鍖栨紡鏂楀喅绛栵級 + # 鈶 best 1fps 鐗堬紙鍏煎鏃х敤閫旓級 + try: + self.finalize_multiframe_mp4() + except Exception as e: + LOGGER.warning("[LIVE] multiframe mp4 finalize failed: %s", e) + try: + self.finalize_mp4() + except Exception as e: + LOGGER.warning("[LIVE] mp4 finalize failed: %s", e) + + # ------------------------------------------------------------ + # Inputs + # ------------------------------------------------------------ + + def feed_user_raw(self, pcm_f32: np.ndarray,t_override: Optional[float] = None) -> None: + if (not self.enabled or not self._started.is_set() + or self._stopping.is_set() or self._t0 is None): + return + if pcm_f32 is None or pcm_f32.size == 0: + return + if pcm_f32.dtype != np.float32: + pcm_f32 = pcm_f32.astype(np.float32) + # 鏂板:鍏佽璋冪敤鑰呮彁渚涚簿纭殑闊抽鏃堕棿杞 t 鎴(rerun 鍦烘櫙); + # 鍚﹀垯鎸 wall clock 鏉(ESP32 鍦烘櫙) + if t_override is not None: + t = max(0.0, t_override) + else: + t = max(0.0, time.monotonic() - self._t0) + #t = max(0.0, time.monotonic() - self._t0) + with self._lock: + if self._user_t_start is None: + self._user_t_start = t + rms = float(np.sqrt(np.mean(pcm_f32 ** 2))) + LOGGER.info("[LIVE] first feed_user_raw(): %d samples rms=%.4f t=%.2fs", + pcm_f32.size, rms, t) + self._user_chunks.append((t, pcm_f32.copy())) + self._user_calls += 1 + + def on_frame(self, jpeg_bytes: Optional[bytes], chunk_idx: int) -> None: + if (not self.enabled or not self._started.is_set() + or self._stopping.is_set() or not jpeg_bytes + or self._t0 is None or self.dir is None): + return + t = time.monotonic() - self._t0 + rel = f"images/img_{chunk_idx:05d}.jpg" + # 瀹為檯鍐 jpg 钀界洏锛堜緵 rerun 鐨 LocalImageSource 璇伙級銆 + try: + img_dir = self.dir / "images" + img_dir.mkdir(parents=True, exist_ok=True) + if not self.no_media: + (self.dir / rel).write_bytes(jpeg_bytes) + except Exception as e: + LOGGER.warning("[LIVE] write frame jpg err: %r", e) + return + with self._lock: + self._frames.append((t, rel, int(chunk_idx))) + + def on_funnel_round(self, decision, chunk_idx: int, text: str = "") -> None: + """瀛樻紡鏂椾竴杞細鏁寸皣澶氬抚(绛涘墠鍘熷)钀界洏 q{NNN}_f{MM}.jpg + frames.jsonl 璁颁竴鏉° + 缁熶竴 record 鐨勬牳蹇冿細涓浠 record 渚涗笁绉 rerun鈥斺 + 鏃犳紡鏂 rerun(姣忚疆鍙栦唬琛ㄥ抚)/ funnel rerun(鏁寸皣閲嶅垽)/ -o rerun(鍙栧綋鏃禸est)銆 + """ + if (not self.enabled or not self._started.is_set() + or self._stopping.is_set() or self._t0 is None or self.dir is None): + return + t = time.monotonic() - self._t0 + frames = getattr(decision, "frames", None) + if not frames: + b = getattr(decision, "best", None) or getattr(decision, "best_jpg", None) + frames = [b] if b else [] + # qid 鐢ㄧ嫭绔嬭疆娆¤鏁帮紙姣忚疆 +1锛夛紝涓嶈兘鐢 latest_frame.sequence鈥斺 + # 閭d釜鍙湪 send 鏃堕掑锛屼竴鐩 reject锛堟檭鍔/绯婏級鏃朵笉鍙 鈫 qid 涓嶅彉 鈫 鏁寸皣 jpg + # 鍙嶅鐢ㄥ悓鍚 q{鍚屽紏_fNN 瑕嗙洊鏃у浘锛宺ecord 瀛樹笉涓嬨乺erun 娌″浘銆 + self._round_counter = getattr(self, "_round_counter", -1) + 1 + qid = self._round_counter + frame_names = [] + try: + img_dir = self.dir / "images" + img_dir.mkdir(parents=True, exist_ok=True) + for i, jpg in enumerate(frames): + if not jpg: + continue + name = f"q{qid:05d}_f{i:02d}.jpg" + if not self.no_media: + (img_dir / name).write_bytes(jpg) + frame_names.append(name) + except Exception as e: + LOGGER.warning("[LIVE] on_funnel_round write err: %r", e) + return + rec = { + "qid": qid, + "t": round(t, 3), + "chunk_idx": int(chunk_idx), + "frames": frame_names, + "best_index": getattr(decision, "best_index", None), + "send": bool(getattr(decision, "send", False)), + "reason": getattr(decision, "reason", ""), + "text": text, + } + with self._lock: + self._funnel_rounds.append(rec) + + # ------------------------------------------------------------ + # WAV writing + # ------------------------------------------------------------ + + @staticmethod + def _nonzero_ratio(chunks: list[tuple[float, np.ndarray]]) -> float: + if not chunks: + return 0.0 + total, nz = 0, 0 + for _, pcm in chunks: + total += pcm.size + nz += int(np.sum(np.abs(pcm) > 1e-4)) + return (nz / total) if total > 0 else 0.0 + + @staticmethod + def _write_track_wav( + path: Path, + t_start: Optional[float], # 浠呯敤浜庢棩蹇楀吋瀹,涓嶅啀鍐冲畾璧风偣 + chunks: list[tuple[float, np.ndarray]], + sr: int, + wall_dur: float, + ) -> float: + """褰曞睆寮忓啓鐩:姣忎釜 chunk 钀藉湪瀹冪湡瀹炵殑 arrival 鏃堕棿涓,绌洪殭鐣 0銆""" + if not chunks: + n_total = max(int(wall_dur * sr), 1) + track = np.zeros(n_total, dtype=np.float32) + audio_dur = 0.0 + else: + # 璁$畻鎬婚暱 = max(鏈鍚庝竴娈靛熬宸, wall_dur) + end_time = max(t + pcm.size / sr for t, pcm in chunks) + n_total = max(int(max(end_time, wall_dur) * sr), 1) + track = np.zeros(n_total, dtype=np.float32) + audio_dur = 0.0 # 杩欓噷鏀逛负"鏈夋晥鏍锋湰"鐨勭疮璁,涓嶅啀鏄 concat 闀垮害 + prev_end_samp = 0 + for t, pcm in chunks: + i = max(0, int(t * sr)) + # 闃叉涓嬩竴娈电殑 arrival time 灏忎簬涓婁竴娈电粨鏉熲斺旈噸鍙犳椂鎸変笂涓娈靛熬宸撮『寤 + i = max(i, prev_end_samp) + j = min(i + pcm.size, n_total) + if j > i: + track[i:j] = pcm[: j - i] + prev_end_samp = j + audio_dur += (j - i) / sr + + i16 = np.clip(track * 32768.0, -32768, 32767).astype(np.int16) + with wave.open(str(path), "wb") as w: + w.setnchannels(1); + w.setsampwidth(2); + w.setframerate(sr) + w.writeframes(i16.tobytes()) + return audio_dur + + # ------------------------------------------------------------ + # MP4 finalize + # ------------------------------------------------------------ + + def finalize_mp4(self) -> Optional[Path]: + if self.no_media: + return None + if not self.enabled or self.dir is None: + return None + if shutil.which("ffmpeg") is None: + LOGGER.warning("[LIVE] ffmpeg not found. WAV/甯у凡淇濆瓨鍦 %s銆", self.dir) + return None + main_out = None + + + try: + main_out = self._do_finalize_mp4() + except Exception as e: + LOGGER.warning("[LIVE] mp4 finalize failed: %s", e) + # 棰濆鍚堟垚涓浠 user-only,鐢ㄤ簬 gateway 8006 瑙嗛杈撳叆娴嬭瘯 + try: + self._do_finalize_useronly_mp4() + except Exception as e: + LOGGER.warning("[LIVE] useronly mp4 finalize failed: %s", e) + return main_out + + def _do_finalize_useronly_mp4(self) -> Optional[Path]: + """棰濆鍚堟垚涓浠藉彧鍚 user 闊宠建鐨 mp4,鐢ㄤ簬缁 gateway 鐨 8006 瑙嗛杈撳叆绔仛 fixture 娴嬭瘯銆""" + d = self.dir + assert d is not None + user_wav = d / "live_user.wav" + if not user_wav.exists(): + LOGGER.info("[LIVE] no user wav, skip useronly mp4") + return None + + with wave.open(str(user_wav), "rb") as w: + u_dur = w.getnframes() / w.getframerate() + if u_dur <= 0.5: + LOGGER.info("[LIVE] user too short (%.2fs), skip useronly mp4", u_dur) + return None + + with self._lock: + frames = list(self._frames) + + concat_txt = d / "_useronly_frames.txt" + if frames: + with concat_txt.open("w", encoding="utf-8") as f: + f.write("ffconcat version 1.0\n") + first_t = frames[0][0] + if first_t > 0.05: + f.write(f"file '{frames[0][1]}'\n") + f.write(f"duration {first_t:.3f}\n") + for i, (rt, p, _idx) in enumerate(frames): + f.write(f"file '{p}'\n") + if i + 1 < len(frames): + dur = max(0.04, frames[i + 1][0] - rt) + else: + dur = max(0.5, u_dur - rt) + f.write(f"duration {dur:.3f}\n") + f.write(f"file '{frames[-1][1]}'\n") + + if not frames: + out = d / "live_useronly.m4a" + cmd = [ + "ffmpeg", "-y", "-hide_banner", "-loglevel", "warning", + "-i", str(user_wav), + "-c:a", "aac", "-b:a", "192k", + "-ac", "1", + "-t", f"{u_dur:.3f}", + str(out), + ] + else: + out = d / "live_useronly.mp4" + cmd = [ + "ffmpeg", "-y", "-hide_banner", "-loglevel", "warning", + "-f", "concat", "-safe", "0", "-i", str(concat_txt), + "-i", str(user_wav), + "-map", "0:v", "-map", "1:a", + "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-pix_fmt", "yuv420p", + "-vsync", "vfr", + "-c:a", "aac", "-b:a", "192k", + "-ac", "1", + "-t", f"{u_dur:.3f}", + str(out), + ] + + LOGGER.info( + "[LIVE] assembling useronly: frames=%d u_dur=%.1fs 鈫 %s", + len(frames), u_dur, out.name, + ) + try: + r = subprocess.run(cmd, capture_output=True, text=True, timeout=600) + if r.returncode != 0: + LOGGER.warning("[LIVE] useronly ffmpeg rc=%d stderr tail:\n%s", + r.returncode, r.stderr[-1200:]) + return None + if out.exists(): + LOGGER.info("[LIVE] 鉁 saved %s (%.1f MB, %.1fs)", + out, out.stat().st_size / 1e6, u_dur) + return out + except subprocess.TimeoutExpired: + LOGGER.warning("[LIVE] useronly ffmpeg timeout (>10min)") + return None + finally: + try: + if concat_txt.exists(): + concat_txt.unlink() + except Exception: + pass + def _do_finalize_mp4(self) -> Optional[Path]: + d = self.dir + assert d is not None + user_wav = d / "live_user.wav" + ai_wav = d / "live_ai.wav" + if not (user_wav.exists() and ai_wav.exists()): + LOGGER.info("[LIVE] no wav files, skip mp4") + return None + + with self._lock: + frames = list(self._frames) + + with wave.open(str(user_wav), "rb") as w: + u_dur = w.getnframes() / w.getframerate() + with wave.open(str(ai_wav), "rb") as w: + a_dur = w.getnframes() / w.getframerate() + total_dur = max(u_dur, a_dur) + if total_dur <= 0.5: + LOGGER.info("[LIVE] session too short (%.2fs), skip mp4", total_dur) + return None + + concat_txt = d / "_live_frames.txt" + if frames: + with concat_txt.open("w", encoding="utf-8") as f: + f.write("ffconcat version 1.0\n") + first_t = frames[0][0] + if first_t > 0.05: + f.write(f"file '{frames[0][1]}'\n") + f.write(f"duration {first_t:.3f}\n") + for i, (rt, p, _idx) in enumerate(frames): + f.write(f"file '{p}'\n") + if i + 1 < len(frames): + dur = max(0.04, frames[i + 1][0] - rt) + else: + dur = max(0.5, total_dur - rt) + f.write(f"duration {dur:.3f}\n") + f.write(f"file '{frames[-1][1]}'\n") + + # apad + -t 璁╃煭鐨勪竴杞ㄨ嚜鍔ㄨˉ灏鹃潤闊冲榻愬埌 mp4 鎬绘椂闀 + if not frames: + out = d / "live_session.m4a" + cmd = [ + "ffmpeg", "-y", "-hide_banner", "-loglevel", "warning", + "-i", str(user_wav), "-i", str(ai_wav), + "-filter_complex", + "[0:a]aresample=24000,aformat=channel_layouts=mono,apad[u];" + "[1:a]aformat=channel_layouts=mono,apad[a];" + "[u][a]amerge=inputs=2[aout]", + "-map", "[aout]", + "-c:a", "aac", "-b:a", "192k", + "-t", f"{total_dur:.3f}", + str(out), + ] + else: + out = d / "live_session.mp4" + cmd = [ + "ffmpeg", "-y", "-hide_banner", "-loglevel", "warning", + "-f", "concat", "-safe", "0", "-i", str(concat_txt), + "-i", str(user_wav), "-i", str(ai_wav), + "-filter_complex", + "[1:a]aresample=24000,aformat=channel_layouts=mono,apad[u];" + "[2:a]aformat=channel_layouts=mono,apad[a];" + "[u][a]amerge=inputs=2[aout]", + "-map", "0:v", "-map", "[aout]", + #"-c:v", "libx264", "-preset", "veryfast", "-pix_fmt", "yuv420p", + "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-pix_fmt", "yuv420p", + "-vsync", "vfr", + "-c:a", "aac", "-b:a", "192k", + "-t", f"{total_dur:.3f}", + str(out), + ] + + LOGGER.info( + "[LIVE] assembling: frames=%d total=%.1fs (u=%.1fs a=%.1fs) 鈫 %s", + len(frames), total_dur, u_dur, a_dur, out.name, + ) + try: + r = subprocess.run(cmd, capture_output=True, text=True, timeout=600) + if r.returncode != 0: + LOGGER.warning( + "[LIVE] ffmpeg rc=%d stderr tail:\n%s", + r.returncode, r.stderr[-1200:], + ) + return None + if out.exists(): + LOGGER.info( + "[LIVE] 鉁 saved %s (%.1f MB, %.1fs)", + out, out.stat().st_size / 1e6, total_dur, + ) + return out + except subprocess.TimeoutExpired: + LOGGER.warning("[LIVE] ffmpeg timeout (>10min)") + return None + finally: + try: + if concat_txt.exists(): + concat_txt.unlink() + except Exception: + pass + + + # ------------------------------------------------------------ + # 鏁寸皣澶氬抚 mp4锛堟紨绀虹敤锛夛細鐢ㄦ瘡杞暣绨囧甯ф嫾锛屾瘮 1fps best 娴佺晠锛 + # 瑙掕惤鐢 ass 瀛楀箷鏍囨敞姣忚疆 send/reject + reason锛岀洿鎺ュ彲瑙嗗寲婕忔枟鍐崇瓥銆 + # 瀵规瘮鏃犳紡鏂/鏈夋紡鏂椾袱浠 mp4锛屽嵆鍙瘉鏄"妯$硦鍥捐鎷掋佷笉杩涙ā鍨"鐨勬纭с + # ------------------------------------------------------------ + def _ass_escape(self, s: str) -> str: + return (s or "").replace("\\", "\\\\").replace("{", "(").replace("}", ")") + + def _gen_reject_ass(self, rounds, total_dur: float) -> Optional[Path]: + """鐢熸垚瑙掕惤鏍囨敞瀛楀箷锛氭瘡杞椂闂存鏄剧ず SEND / REJECT:reason銆""" + d = self.dir + if d is None or not rounds: + return None + def _fmt(t): + t = max(0.0, t) + h = int(t // 3600); m = int((t % 3600) // 60) + s = t % 60 + return f"{h:d}:{m:02d}:{s:05.2f}" + ass = d / "_reject_notes.ass" + header = ( + "[Script Info]\nScriptType: v4.00+\nPlayResX: 1280\nPlayResY: 720\n\n" + "[V4+ Styles]\n" + "Format: Name, Fontname, Fontsize, PrimaryColour, Bold, Alignment, " + "MarginL, MarginR, MarginV, BorderStyle, Outline, Shadow\n" + # Alignment 9 = 鍙充笂瑙掞紱绾=&H000000FF 缁=&H0000FF00 (ASS 鏄 &HAABBGGRR) + "Style: REJ,Arial,36,&H000000FF,1,9,20,20,20,1,2,0\n" + "Style: SND,Arial,36,&H0000FF00,1,9,20,20,20,1,2,0\n\n" + "[Events]\nFormat: Layer, Start, End, Style, Text\n" + ) + lines = [] + for i, rec in enumerate(rounds): + t0 = float(rec.get("t", i)) + t1 = float(rounds[i + 1].get("t", t0 + 1.0)) if i + 1 < len(rounds) else total_dur + if t1 <= t0: + t1 = t0 + 0.3 + send = bool(rec.get("send")) + reason = self._ass_escape(rec.get("reason", "")) + if send: + style, txt = "SND", "SEND 鉁" + else: + style, txt = "REJ", f"REJECT 鉁 {reason}" + lines.append(f"Dialogue: 0,{_fmt(t0)},{_fmt(t1)},{style},,{txt}") + try: + ass.write_text(header + "\n".join(lines) + "\n", encoding="utf-8") + return ass + except Exception as e: + LOGGER.warning("[LIVE] write reject ass err: %r", e) + return None + + def finalize_multiframe_mp4(self) -> Optional[Path]: + if self.no_media: + return None + """鏁寸皣澶氬抚 mp4 + 瑙掕惤 reject 鏍囨敞銆傛紨绀虹敤锛屾瘮 best 1fps 娴佺晠銆""" + if not self.enabled or self.dir is None: + return None + if shutil.which("ffmpeg") is None: + LOGGER.warning("[LIVE] 鈿 ffmpeg 鏈畨瑁咃紝鏃犳硶鐢熸垚 mp4锛" + "瑁呮硶: conda install -c conda-forge ffmpeg銆" + "甯у拰 wav 宸插瓨鍦 %s锛屽彲鎵嬪姩鎷笺", self.dir) + return None + d = self.dir + user_wav = d / "live_user.wav" + ai_wav = d / "live_ai.wav" + if not (user_wav.exists() and ai_wav.exists()): + return None + with self._lock: + rounds = list(self._funnel_rounds) + if not rounds: + LOGGER.info("[LIVE] no funnel rounds, skip multiframe mp4") + return None + + with wave.open(str(user_wav), "rb") as w: + u_dur = w.getnframes() / w.getframerate() + with wave.open(str(ai_wav), "rb") as w: + a_dur = w.getnframes() / w.getframerate() + total_dur = max(u_dur, a_dur) + if total_dur <= 0.5: + return None + + # 鏁寸皣澶氬抚 concat锛氭瘡杞殑 t 鍒颁笅涓杞殑 t 涔嬮棿锛屽潎鍒嗙粰璇ヨ疆鐨 N 甯э紙楂 fps锛 + concat_txt = d / "_multiframe.txt" + with concat_txt.open("w", encoding="utf-8") as f: + f.write("ffconcat version 1.0\n") + first_t = float(rounds[0].get("t", 0.0)) + if first_t > 0.05 and rounds[0].get("frames"): + f.write(f"file 'images/{Path(rounds[0]['frames'][0]).name}'\n") + f.write(f"duration {first_t:.3f}\n") + for i, rec in enumerate(rounds): + names = rec.get("frames") or [] + if not names: + continue + t0 = float(rec.get("t", i)) + t1 = float(rounds[i + 1].get("t", t0 + 1.0)) if i + 1 < len(rounds) else total_dur + round_dur = max(0.12, t1 - t0) + per = round_dur / len(names) # 鏁寸皣鍐呭甯у潎鍒 鈫 楂 fps + for nm in names: + f.write(f"file 'images/{Path(nm).name}'\n") + f.write(f"duration {max(0.03, per):.3f}\n") + # 鏈抚鍏滃簳 + last_names = rounds[-1].get("frames") or [] + if last_names: + f.write(f"file 'images/{Path(last_names[-1]).name}'\n") + + ass = self._gen_reject_ass(rounds, total_dur) + vf = "format=yuv420p" + if ass is not None: + safe = str(ass.name) # 鐩稿 cwd 闇鍦 dir 涓嬭繍琛岋紱鐢ㄧ粷瀵规洿绋 + abs_ass = str(ass.resolve()).replace("\\", "/").replace(":", "\\:") + vf = f"ass='{abs_ass}',format=yuv420p" + + out = d / "live_multiframe.mp4" + cmd = [ + "ffmpeg", "-y", "-hide_banner", "-loglevel", "warning", + "-f", "concat", "-safe", "0", "-i", str(concat_txt), + "-i", str(user_wav), "-i", str(ai_wav), + "-filter_complex", + "[1:a]aresample=24000,aformat=channel_layouts=mono,apad[u];" + "[2:a]aformat=channel_layouts=mono,apad[a];" + "[u][a]amerge=inputs=2[aout]", + "-map", "0:v", "-map", "[aout]", + "-vf", vf, + "-c:v", "libx264", "-preset", "medium", "-crf", "20", "-pix_fmt", "yuv420p", + "-vsync", "vfr", + "-c:a", "aac", "-b:a", "192k", + "-t", f"{total_dur:.3f}", + str(out), + ] + LOGGER.info("[LIVE] assembling multiframe mp4: rounds=%d total=%.1fs 鈫 %s", + len(rounds), total_dur, out.name) + try: + r = subprocess.run(cmd, capture_output=True, text=True, timeout=600) + if r.returncode != 0: + LOGGER.warning("[LIVE] multiframe ffmpeg rc=%d stderr:\n%s", + r.returncode, r.stderr[-1200:]) + return None + if out.exists(): + LOGGER.info("[LIVE] 鉁 multiframe mp4 saved %s (%.1f MB)", + out, out.stat().st_size / 1e6) + return out + except subprocess.TimeoutExpired: + LOGGER.warning("[LIVE] multiframe ffmpeg timeout") + return None + finally: + for p in (concat_txt, ass): + try: + if p is not None and p.exists(): + p.unlink() + except Exception: + pass diff --git a/extensions/assistive_harness/phase_b/requirements-phase-b.txt b/extensions/assistive_harness/phase_b/requirements-phase-b.txt new file mode 100644 index 0000000..4cd81f4 --- /dev/null +++ b/extensions/assistive_harness/phase_b/requirements-phase-b.txt @@ -0,0 +1,34 @@ +# OpenGlass 鈥 extensions/assistive_harness 渚濊禆锛堚憽鈶⑩懀 閾捐矾锛氳闊虫帶鍒 + CV 婕忔枟锛 +# +# 瑁呭湪**鍜 MiniCPM-o-Demo 鍚屼竴涓 conda 鐜**閲岋細panel 璧峰瓙杩涚▼鏃剁户鎵垮惎鍔ㄥ畠鐨勭幆澧冿紝 +# 鑰 worker/gateway 涔熷湪閭d釜鐜璺戙備笉瑕佸彟寤虹幆澧冦 +# +# conda activate <浣犺窇 MiniCPM-o-Demo 鐨勭幆澧> +# pip install -r requirements-phase-b.txt + +# 鈹鈹 鍩虹 鈹鈹 +aiohttp>=3.13 # gateway/harness 鐨 WebSocket 瀹㈡埛绔 + bridge_ui 鏈嶅姟 +numpy>=1.24 # 闊抽/鍥惧儚鏁扮粍 +Pillow>=10 # --rotate 杞锛圥IL锛 +sounddevice>=0.5 # PC 鎵0鍣ㄦ挱鏀撅紙AI 璇煶銆乺eject 鎻愮ず闊筹級 +requests>=2.31 # ESP32 HTTP 鎺у埗锛堝垎杈ㄧ巼/瀵圭劍锛 +PyYAML>=6.0 # harness 鐨 skills 閰嶇疆 + +# 鈹鈹 8021 harness锛堣闊虫帶鍒惰吙锛屸憽鈶⑩懀 闇瑕侊級鈹鈹 +fastapi>=0.110 +uvicorn>=0.27 +funasr>=1.3 # 娴佸紡 ASR銆傛ā鍨嬪彟澶栦笅杞斤紝璺緞鍐欒繘 runtime.local.json +cryptography>=42 # panel 棣栨鍚姩鑷璇佷功锛堟病鏈夊畠浼氶鍥炶皟 openssl锛 + +# 鈹鈹 CV 婕忔枟锛堚憿鈶 闇瑕侊級鈹鈹 +opencv-python>=4.8 # 娓呮櫚搴/鍏夋祦/浜害鍒ゆ嵁 +# +# 鈽 鏂瑰悜鍒嗙被鍣紙orient_flipped / orient_sideways 鍒ゆ嵁锛夆斺 +# 瑁呬笉涓婃垨鐗堟湰涓嶅鏃朵細**闈欓粯鍥為鍒版棭鏈熺殑绠鏄撳垽鎹**锛岃〃鐜版槸鐢婚潰鏄庢槑鏄鐨 +# 鍗翠竴鐩存姤"鐢婚潰濂藉儚鍙嶄簡"銆傚惎鍔ㄦ棩蹇楅噷浼氭湁锛 +# [棰勭儹] !! PaddleOCR 鏂瑰悜鍒嗙被鍣ㄥ姞杞藉け璐ワ紝宸插洖閫鍒拌交閲 CV 鍒ゆ嵁 ... +# 鐪嬪埌杩欒灏辫鏄庝笅闈袱涓病瑁呭ソ锛屽埆蹇界暐銆 +# +# 瀹炴祴鍙敤缁勫悎锛圕PU 鐗堝嵆鍙紝妯″瀷鍙湁 6.75MB锛夛細 +paddlepaddle>=3.0 # GPU 鏈哄櫒鍙崲 paddlepaddle-gpu锛屼絾 CPU 鐗堣冻澶 +paddleocr>=3.0 # 鍙敤鍏朵腑鐨 DocImgOrientationClassification diff --git a/extensions/assistive_harness/phase_b/rerun_source.py b/extensions/assistive_harness/phase_b/rerun_source.py new file mode 100644 index 0000000..448862d --- /dev/null +++ b/extensions/assistive_harness/phase_b/rerun_source.py @@ -0,0 +1,368 @@ +"""rerun_source.py 鈥 鐢ㄥ凡褰曞埗 session 褰撹緭鍏ラ噸璺戞ā鍨 + +鏇挎崲涓昏剼鏈噷鐨勪袱涓綉缁滆緭鍏ュ嚱鏁帮細 + - esp32_audio_reader 鈫 local_pcm_reader + - esp32_capture_image 鈫 LocalImageSource.capture +""" + +from __future__ import annotations + +import asyncio +import json +import logging +import time +from pathlib import Path +from typing import Optional + +import numpy as np +import sounddevice as sd +import threading + + +LOGGER = logging.getLogger("rerun_source") + +_PACKET_MS = 40 +_PACKET_SAMPLES = 640 +_SAMPLE_RATE = 16000 + + +class _UserMonitor: + """鐙珛 PortAudio OutputStream,鐢ㄤ簬 rerun 鏃舵挱鏀 user 鐩戝惉闊抽銆 + + 涓轰粈涔堜笉澶嶇敤涓 SpeakerPlayer:涓 speaker 鏄崟 FIFO ring, + AI 鍜 user 涓よ矾 enqueue 鍏变韩鍚屼竴涓 ring 鏃,DAC 鎸夊叆闃熼『搴忔秷璐逛細 + 瀵艰嚧 AI 闊抽琚 user 闊抽"鎻掗槦"鈥斺斿惉鎰熸槸 AI 杈撳嚭鍗¢】銆 + + 鐙珛 OutputStream 鐢 OS 绔贩闊(鍚屼竴澹板崱鏃堕挓椹卞姩),鐗╃悊涓婁笉浜掔浉鎸ゅ帇銆 + """ + + def __init__(self, sample_rate: int = 24000, ring_capacity_s: float = 2.0): + self.sr = int(sample_rate) + self.ring_cap = int(ring_capacity_s * self.sr) + self._buf = np.zeros(self.ring_cap, dtype=np.float32) + self._r = 0 + self._w = 0 + self._size = 0 + self._lock = threading.Lock() + self._dropped = 0 + try: + self._stream = sd.OutputStream( + samplerate=self.sr, channels=1, dtype="float32", + blocksize=0, callback=self._cb, + ) + self._stream.start() + LOGGER.info("[RERUN-MON] user monitor stream started: sr=%d ring=%.1fs", + self.sr, ring_capacity_s) + except Exception as e: + LOGGER.warning("[RERUN-MON] failed to open OutputStream: %r", e) + self._stream = None + + def _cb(self, outdata, frames, time_info, status): + with self._lock: + avail = min(frames, self._size) + if avail > 0: + first = min(avail, self.ring_cap - self._r) + outdata[:first, 0] = self._buf[self._r:self._r + first] + if avail > first: + outdata[first:avail, 0] = self._buf[:avail - first] + self._r = (self._r + avail) % self.ring_cap + self._size -= avail + if avail < frames: + outdata[avail:, 0] = 0.0 + + def enqueue(self, pcm_f32: np.ndarray) -> None: + if self._stream is None or pcm_f32 is None or pcm_f32.size == 0: + return + if pcm_f32.dtype != np.float32: + pcm_f32 = pcm_f32.astype(np.float32) + n = pcm_f32.size + if n > self.ring_cap: + pcm_f32 = pcm_f32[-self.ring_cap:] + n = pcm_f32.size + with self._lock: + if self._size + n > self.ring_cap: + drop = (self._size + n) - self.ring_cap + self._r = (self._r + drop) % self.ring_cap + self._size -= drop + self._dropped += drop + first = min(n, self.ring_cap - self._w) + self._buf[self._w:self._w + first] = pcm_f32[:first] + if n > first: + self._buf[:n - first] = pcm_f32[first:n] + self._w = (self._w + n) % self.ring_cap + self._size += n + + def stop(self) -> None: + if self._stream is None: + return + try: + self._stream.stop() + self._stream.close() + except Exception as e: + LOGGER.warning("[RERUN-MON] stop err: %r", e) + finally: + self._stream = None + LOGGER.info("[RERUN-MON] user monitor stopped (dropped=%d samples)", + self._dropped) + + + +async def local_pcm_reader( + pcm_path: Path, + ring, + stop_evt: asyncio.Event, + stats, + live_rec=None, + speaker=None, # 鈫 鏂板 + speaker_sr: int = 24000, # 鈫 鏂板:speaker 鏈熸湜鐨勯噰鏍风巼 + speed: float = 1.0, + drain_s: float = 5.0, + prefill_s: float = 0.0, +) -> None: + pcm_path = Path(pcm_path) + if not pcm_path.exists(): + LOGGER.error("[RERUN] pcm not found: %s", pcm_path) + stop_evt.set() + return + + # 浼樺厛鐢 live_user.wav (LiveRecorder 鐩存帴鍐欑殑,骞插噣); + # 鍥為鐢 user_raw.pcm (涓诲惊鐜 ring.slice 钀界洏鐨,甯 40% 闆跺~鍏) + import wave + wav_path = pcm_path.parent / "live_user.wav" + pcm_i16 = None + src_label = "" + if wav_path.exists(): + try: + with wave.open(str(wav_path), "rb") as w: + sr_in = w.getframerate() + nch = w.getnchannels() + sw = w.getsampwidth() + if sr_in == _SAMPLE_RATE and nch == 1 and sw == 2: + pcm_i16 = np.frombuffer(w.readframes(w.getnframes()), dtype=np.int16) + src_label = f"live_user.wav (clean, {sr_in}Hz mono int16)" + else: + LOGGER.warning( + "[RERUN] live_user.wav format unexpected (sr=%d ch=%d sw=%d), " + "fallback to user_raw.pcm", sr_in, nch, sw, + ) + except Exception as e: + LOGGER.warning("[RERUN] failed to read live_user.wav: %r, fallback to user_raw.pcm", e) + pcm_i16 = None + + if pcm_i16 is None: + raw = pcm_path.read_bytes() + pcm_i16 = np.frombuffer(raw, dtype=np.int16) + src_label = "user_raw.pcm (may contain ring.slice zero-padding)" + + total_samples = pcm_i16.size + total_dur = total_samples / _SAMPLE_RATE + LOGGER.info("[RERUN] audio source: %s", src_label) + # rerun 鐢ㄦ埛鐩戝惉:鐙珛 OutputStream,閬垮厤鍜 AI 鍏辩敤 ring 瀵艰嚧浜掔浉鍗¢】 + user_mon = None + if speaker is not None: + try: + user_mon = _UserMonitor(sample_rate=speaker_sr) + except Exception as e: + LOGGER.warning("[RERUN] failed to create user monitor: %r", e) + user_mon = None + + + LOGGER.info( + "[RERUN] pcm loaded: %s samples=%d dur=%.2fs speed=%.2fx prefill=%.1fs", + pcm_path, total_samples, total_dur, speed, prefill_s, + ) + + packet_interval = (_PACKET_MS / 1000.0) / max(speed, 0.01) + ts_ms = 0 + cursor = 0 + pkt_count = 0 + prefill_packets = int(prefill_s * 1000 / _PACKET_MS) + next_tick = time.monotonic() # 鈫 鏀:涓嶅啀 + prefill_s + audio_t_start = None + while cursor < total_samples and not stop_evt.is_set(): + end = min(cursor + _PACKET_SAMPLES, total_samples) + chunk_i16 = pcm_i16[cursor:end] + cursor = end + pkt_count += 1 + + try: + stats["rx_pkts"] = pkt_count + stats["esp_drops"] = 0 + except Exception: + pass + + try: + await ring.write(ts_ms, chunk_i16) + except Exception as e: + LOGGER.warning("[RERUN] ring.write err: %r", e) + + if live_rec is not None: + try: + # 鈫 鍏抽敭鏀瑰姩:鐢"绱闊抽鏃堕暱"浣滀负 t 鎴,鑰屼笉鏄 wall clock + # 杩欐牱 LiveRecorder 鍐欑洏鏃 t * sr 绠楀嚭鏉ョ殑 i 涓 pcm.size 瀹岀編瀵归綈 + if audio_t_start is None: + audio_t_start = time.monotonic() - live_rec._t0 if live_rec._t0 else 0.0 + audio_t = audio_t_start + ts_ms / 1000.0 + live_rec.feed_user_raw( + chunk_i16.astype(np.float32) / 32768.0, + t_override=audio_t, + ) + # rerun 妯″紡:鎶 user 闊抽涔熸帹缁欐壃澹板櫒,杩欐牱鑳藉惉鍒板師褰曞埗鐨勭敤鎴峰湪璇翠粈涔 + # (live 妯″紡涓嶈繖涔堝仛,鍥犱负鐢ㄦ埛鏈汉灏卞湪閭h璇,浼氬拰鎵0鍣ㄥ彔鍔犮 + # rerun 鏄簨鍚庡洖鏀,娌″洖澹伴闄,绛夊悓 8006 娴忚鍣ㄦ挱 mp4 鐨勪綋楠) + if user_mon is not None: + try: + chunk_f32 = chunk_i16.astype(np.float32) / 32768.0 + if speaker_sr != _SAMPLE_RATE: + ratio = speaker_sr / _SAMPLE_RATE + new_len = int(chunk_f32.size * ratio) + x_old = np.linspace(0, 1, chunk_f32.size, endpoint=False) + x_new = np.linspace(0, 1, new_len, endpoint=False) + chunk_resampled = np.interp(x_new, x_old, chunk_f32).astype(np.float32) + else: + chunk_resampled = chunk_f32 + # 鍏抽敭:璧板師濮 enqueue,涓嶈蛋 LiveRecorder hook + # (hook 浼氭妸杩欐潯鐢ㄦ埛闊抽閿欒鍦扮疮绉繘 _ai_chunks, + # 瀵艰嚧 mp4 鐨 ai 杞ㄩ噷涔熸湁鐢ㄦ埛澹伴煶,鍚捣鏉ュ儚"鐢ㄦ埛琚綍浜嗕袱娆") + user_mon.enqueue(chunk_resampled) + except Exception as e: + LOGGER.warning("[RERUN] speaker.enqueue err: %r", e) + + #live_rec.feed_user_raw(chunk_i16.astype(np.float32) / 32768.0) + except Exception as e: + LOGGER.warning("[RERUN] live_rec.feed_user_raw err: %r", e) + + ts_ms += _PACKET_MS + + # prefill 闃舵:璁╁嚭浜嬩欢寰幆浣嗕笉鐪熺潯 + if pkt_count <= prefill_packets: + await asyncio.sleep(0) + continue + + # prefill 鈫 realtime 鍒囨崲閭d竴鍒,閲嶇疆鍩虹嚎 + if pkt_count == prefill_packets + 1: + LOGGER.info( + "[RERUN] prefill done (%d packets, %.2fs audio in ring), " + "switching to realtime pacing", + prefill_packets, prefill_packets * _PACKET_MS / 1000.0, + ) + next_tick = time.monotonic() # 鈫 鍏抽敭:鍩虹嚎閲嶇疆鍒版鍒 + + next_tick += packet_interval + sleep_for = next_tick - time.monotonic() + if sleep_for > 0: + try: + await asyncio.wait_for(stop_evt.wait(), timeout=sleep_for) + break # 琚閮 stop:璺冲嚭鎺ㄦ祦寰幆,浠嶈蛋涓嬮潰鐨 drain(绛夋ā鍨嬭瀹) + except asyncio.TimeoutError: + pass + + try: + stats["audio_done"] = True + stats["audio_done_ts"] = time.monotonic() + except Exception: + pass + + LOGGER.info( + "[RERUN] pcm exhausted: pushed=%d packets (%.2fs audio). draining...", + pkt_count, ts_ms / 1000.0, + ) + + # 闊抽宸叉帹瀹屽苟鏍囪 audio_done銆傜粨鏉熺殑瑁佸喅鏉冧氦缁欎富 send loop: + # 瀹冧細鍦 audio_done 鍚庣瓑妯″瀷璇村畬(杞 last_speak_ts)鍐 set stop_evt銆 + # reader 杩欓噷鍙渶绛夊緟閭d釜淇″彿;甯﹀厹搴曚笂闄愰槻姝㈠紓甯稿崱姝汇 + max_wait_s = 120.0 + waited = 0.0 + while waited < max_wait_s and not stop_evt.is_set(): + try: + await asyncio.wait_for(stop_evt.wait(), timeout=0.2) + break + except asyncio.TimeoutError: + waited += 0.2 + if waited >= max_wait_s: + LOGGER.warning("[RERUN] drain 杈惧埌涓婇檺 %.0fs,寮哄埗缁撴潫", max_wait_s) + + if user_mon is not None: + user_mon.stop() + + if not stop_evt.is_set(): + LOGGER.info("[RERUN] drain done, signaling stop") + stop_evt.set() + + +class LocalImageSource: + def __init__(self, session_dir: Path): + self.session_dir = Path(session_dir) + self.images_dir = self.session_dir / "images" + self.events_path = self.session_dir / "events.jsonl" + + if not self.images_dir.exists(): + raise FileNotFoundError(f"images/ not found: {self.images_dir}") + if not self.events_path.exists(): + raise FileNotFoundError(f"events.jsonl not found: {self.events_path}") + + self._map: dict[int, Optional[str]] = {} + self._build_map() + + n_total = len(self._map) + n_with_img = sum(1 for v in self._map.values() if v) + LOGGER.info( + "[RERUN] image map: total_chunks=%d with_image=%d (missing=%d) from %s", + n_total, n_with_img, n_total - n_with_img, self.events_path.name, + ) + if n_total == 0: + # 鈫 鏂板:璇婃柇绌 events.jsonl + n_files = len(list(self.images_dir.glob("img_*.jpg"))) + LOGGER.warning( + "[RERUN] events.jsonl has NO chunk_sent entries! " + "but images/ has %d files. Falling back to filename-order matching.", + n_files, + ) + self._fallback_filename_order(n_files) + + def _build_map(self) -> None: + with self.events_path.open("r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + evt = json.loads(line) + except Exception: + continue + # 鈫 鏀 1:鐪嬪疄闄 events.jsonl 鐢ㄤ粈涔 kind 瀛楁 + kind = evt.get("kind") or evt.get("type") or "" + if kind != "chunk_sent": + continue + # 鈫 鏀 2:chunk_idx 瀛楁鍚嶅吋瀹 + idx = evt.get("idx", evt.get("chunk_idx")) + img = evt.get("img") + if idx is None: + continue + self._map[int(idx)] = img if img else None + + def _fallback_filename_order(self, n_files: int) -> None: + """events.jsonl 鏄┖鐨勬椂鍊欑殑鍏滃簳:鎸 img_NNNNN.jpg 鏂囦欢鍚嶆槧灏 chunk_idx銆""" + for p in sorted(self.images_dir.glob("img_*.jpg")): + try: + # 鏂囦欢鍚嶅舰濡 img_00007.jpg 鈫 7 + idx = int(p.stem.split("_")[-1]) + self._map[idx] = p.name + except Exception: + pass + LOGGER.info("[RERUN] fallback map built: %d chunks (filename-order)", + len(self._map)) + + async def capture(self, chunk_idx: int) -> Optional[bytes]: + rel = self._map.get(int(chunk_idx)) + if not rel: + return None + name = Path(rel).name + path = self.images_dir / name + if not path.exists(): + return None + try: + return path.read_bytes() + except Exception as e: + LOGGER.warning("[RERUN] image read err: %r", e) + return None \ No newline at end of file diff --git a/extensions/assistive_harness/phase_b/rokid_runtime.py b/extensions/assistive_harness/phase_b/rokid_runtime.py new file mode 100644 index 0000000..af9f8cc --- /dev/null +++ b/extensions/assistive_harness/phase_b/rokid_runtime.py @@ -0,0 +1,1703 @@ +from __future__ import annotations + +import argparse +import asyncio +import base64 +import io +import json +import logging +import os +import signal +import ssl +import threading +import time +import uuid +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Awaitable, Callable, Protocol + +import aiohttp +from aiohttp import web +import numpy as np + +from ..registry import SkillRegistry + +try: + from PIL import Image +except Exception: # pragma: no cover - optional at import time + Image = None + + +LOG = logging.getLogger("assistive_harness.phase_b.rokid") +SAMPLE_RATE_IN = 16_000 +SAMPLE_RATE_OUT = 24_000 +PCM16_WIDTH = 2 +CTRL_C_HARD_EXIT_S = 10.0 + + +def now_ms() -> float: + return time.time() * 1000.0 + + +class CtrlCExitWatchdog: + """Guarantee that a Windows Ctrl+C cannot leave the Rokid port behind.""" + + def __init__(self, hard_exit_s: float = CTRL_C_HARD_EXIT_S): + self.hard_exit_s = max(1.0, float(hard_exit_s)) + self._armed = False + self._lock = threading.Lock() + + @property + def armed(self) -> bool: + return self._armed + + def arm(self) -> None: + with self._lock: + if self._armed: + return + self._armed = True + LOG.info( + "Console interrupt received; closing Rokid listener and runtime " + "(hard cutoff %.1fs)", + self.hard_exit_s, + ) + threading.Thread( + target=self._force_exit_after_deadline, + name="rokid-ctrl-c-watchdog", + daemon=True, + ).start() + + def _force_exit_after_deadline(self) -> None: + time.sleep(self.hard_exit_s) + LOG.error( + "Rokid shutdown exceeded %.1fs; forcing process exit so port 18080 " + "cannot remain occupied", + self.hard_exit_s, + ) + logging.shutdown() + os._exit(130) + + +def pcm16le_to_float32(raw: bytes) -> np.ndarray: + if len(raw) % PCM16_WIDTH: + raw = raw[:-1] + if not raw: + return np.zeros(0, dtype=np.float32) + return ( + np.frombuffer(raw, dtype=" bytes: + """Apply device-specific gain with saturation while preserving PCM16 LE.""" + if gain <= 0: + raise ValueError("input gain must be greater than zero") + if gain == 1.0 or not raw: + return raw + samples = np.frombuffer(raw, dtype=" np.ndarray: + """Return a short, non-speech two-note cue for restart_complete.""" + + amplitude = min(1.0, max(0.0, float(amplitude))) + + def tone(frequency_hz: float, duration_s: float) -> np.ndarray: + count = max(2, int(sample_rate * duration_s)) + phase = np.arange(count, dtype=np.float32) / float(sample_rate) + envelope = np.sin(np.linspace(0.0, np.pi, count, dtype=np.float32)) ** 2 + return ( + amplitude + * envelope + * np.sin(2.0 * np.pi * frequency_hz * phase) + ).astype(np.float32) + + gap = np.zeros(max(1, int(sample_rate * 0.030)), dtype=np.float32) + return np.concatenate((tone(880.0, 0.080), gap, tone(1174.66, 0.105))) + + +def float32_to_base64(samples: np.ndarray) -> str: + return base64.b64encode( + samples.astype(np.float32, copy=False).tobytes() + ).decode("ascii") + + +def same_slots(left: dict[str, Any], right: dict[str, Any]) -> bool: + return {str(key): str(value) for key, value in left.items()} == { + str(key): str(value) for key, value in right.items() + } + + +def rotate_jpeg_clockwise(jpeg: bytes, degrees: int, quality: int = 95) -> bytes: + degrees %= 360 + if degrees == 0 or Image is None: + return jpeg + methods = { + 90: Image.Transpose.ROTATE_270, + 180: Image.Transpose.ROTATE_180, + 270: Image.Transpose.ROTATE_90, + } + method = methods.get(degrees) + if method is None: + raise ValueError("image rotation must be 0, 90, 180, or 270") + with Image.open(io.BytesIO(jpeg)) as image: + output = io.BytesIO() + image.convert("RGB").transpose(method).save( + output, format="JPEG", quality=quality + ) + return output.getvalue() + + +class DropOldestAudioQueue: + """Bounded device-audio queue that always preserves the newest packets.""" + + def __init__(self, max_packets: int): + self._queue: asyncio.Queue[bytes] = asyncio.Queue(maxsize=max(1, max_packets)) + self.dropped_packets = 0 + + @property + def size(self) -> int: + return self._queue.qsize() + + def put_nowait(self, raw: bytes) -> None: + if self._queue.full(): + try: + self._queue.get_nowait() + self.dropped_packets += 1 + except asyncio.QueueEmpty: + pass + self._queue.put_nowait(raw) + + async def get(self, timeout_s: float) -> bytes | None: + try: + return await asyncio.wait_for(self._queue.get(), timeout=timeout_s) + except asyncio.TimeoutError: + return None + + def clear(self) -> int: + cleared = 0 + while True: + try: + self._queue.get_nowait() + cleared += 1 + except asyncio.QueueEmpty: + return cleared + + +class AudioMirrorChunker: + """Repacketize Rokid PCM into the same 100 ms frames used in Phase A.""" + + def __init__(self, samples_per_frame: int = 1_600): + self.target_bytes = max(1, samples_per_frame) * PCM16_WIDTH + self._carry = bytearray() + + def feed(self, raw: bytes) -> list[np.ndarray]: + if len(raw) % PCM16_WIDTH: + raw = raw[:-1] + self._carry.extend(raw) + frames: list[np.ndarray] = [] + while len(self._carry) >= self.target_bytes: + frame = bytes(self._carry[: self.target_bytes]) + del self._carry[: self.target_bytes] + frames.append(pcm16le_to_float32(frame)) + return frames + + def clear(self) -> None: + self._carry.clear() + + +@dataclass(slots=True) +class LatestFrame: + jpeg: bytes | None = None + timestamp_ms: float = 0.0 + sequence: int = 0 + + def set(self, jpeg: bytes, timestamp_ms: float) -> None: + self.jpeg = jpeg + self.timestamp_ms = timestamp_ms + self.sequence += 1 + + +@dataclass(slots=True) +class OutputGate: + generation: int = 0 + current_skill: str = "idle_chat" + current_slots: dict[str, Any] | None = None + speech_hold_active: bool = False + drop_output_until_listen: bool = False + restart_in_progress: bool = False + dropped_old_text: int = 0 + dropped_old_audio: int = 0 + stop_count: int = 0 + local_cue_mute_until_mono: float = 0.0 + + def __post_init__(self) -> None: + if self.current_slots is None: + self.current_slots = {} + + def stop(self) -> None: + self.stop_count += 1 + self.speech_hold_active = True + self.drop_output_until_listen = True + + def resume(self) -> None: + self.speech_hold_active = False + self.drop_output_until_listen = False + + def replacement_ready(self) -> None: + self.speech_hold_active = False + self.drop_output_until_listen = False + self.restart_in_progress = False + + +class SpeakerSink(Protocol): + async def start(self) -> None: ... + + async def enqueue(self, pcm: np.ndarray, generation: int) -> None: ... + + async def block_and_flush(self) -> None: ... + + async def resume(self) -> None: ... + + async def close(self) -> None: ... + + def pending_ms(self) -> float: ... + + +class NullSpeaker: + async def start(self) -> None: + return None + + async def enqueue(self, pcm: np.ndarray, generation: int) -> None: + return None + + async def block_and_flush(self) -> None: + return None + + async def resume(self) -> None: + return None + + async def close(self) -> None: + return None + + def pending_ms(self) -> float: + return 0.0 + + +class PCSpeaker: + """Interruptible 24 kHz PC playback with a short write quantum.""" + + def __init__(self, block_ms: int = 50): + self.block_samples = max(240, int(SAMPLE_RATE_OUT * block_ms / 1000)) + self.queue: asyncio.Queue[tuple[int, int, np.ndarray]] = asyncio.Queue(maxsize=32) + self.blocked = False + self.epoch = 0 + self._stream: Any = None + self._worker: asyncio.Task[None] | None = None + self._io_lock = asyncio.Lock() + self._pending_samples = 0 + + def _open_stream(self) -> None: + import sounddevice as sd # type: ignore + + stream = sd.OutputStream( + samplerate=SAMPLE_RATE_OUT, + channels=1, + dtype="float32", + blocksize=self.block_samples, + latency="low", + ) + stream.start() + self._stream = stream + + def _close_stream(self) -> None: + stream = self._stream + self._stream = None + if stream is None: + return + try: + stream.abort() + finally: + stream.close() + + async def start(self) -> None: + try: + await asyncio.to_thread(self._open_stream) + self._worker = asyncio.create_task(self._play_loop()) + LOG.info("PC speaker output ready") + except Exception as exc: # pragma: no cover - hardware dependent + self._stream = None + LOG.warning("PC speaker disabled: %s", exc) + + async def enqueue(self, pcm: np.ndarray, generation: int) -> None: + if self._stream is None or self.blocked or pcm.size == 0: + return + if self.queue.full(): + try: + _epoch, _generation, dropped = self.queue.get_nowait() + self._pending_samples = max( + 0, self._pending_samples - int(dropped.size) + ) + except asyncio.QueueEmpty: + pass + copied = pcm.astype(np.float32, copy=True) + self._pending_samples += int(copied.size) + self.queue.put_nowait((self.epoch, generation, copied)) + + async def _play_loop(self) -> None: + while True: + epoch, _generation, pcm = await self.queue.get() + try: + for offset in range(0, pcm.size, self.block_samples): + if self.blocked or epoch != self.epoch or self._stream is None: + break + chunk = pcm[offset : offset + self.block_samples].reshape(-1, 1) + stream = self._stream + try: + async with self._io_lock: + if self.blocked or epoch != self.epoch or self._stream is not stream: + break + await asyncio.to_thread(stream.write, chunk) + except asyncio.CancelledError: + raise + except Exception as exc: # pragma: no cover - hardware dependent + LOG.warning("speaker write failed: %s", exc) + break + finally: + self._pending_samples = max( + 0, self._pending_samples - int(pcm.size) + ) + + async def block_and_flush(self) -> None: + self.blocked = True + self.epoch += 1 + self._pending_samples = 0 + while True: + try: + self.queue.get_nowait() + except asyncio.QueueEmpty: + break + if self._stream is not None: + try: + # Windows MME may reject start() immediately after abort() while + # the previous buffer is still retiring. Close the device here + # and create a fresh stream only when output is resumed. + async with self._io_lock: + await asyncio.to_thread(self._close_stream) + except Exception as exc: # pragma: no cover - hardware dependent + LOG.warning("speaker flush failed: %s", exc) + + async def resume(self) -> None: + if self._stream is None: + try: + async with self._io_lock: + if self._stream is None: + await asyncio.to_thread(self._open_stream) + except Exception as exc: # pragma: no cover - hardware dependent + LOG.warning("speaker resume failed: %s", exc) + self.blocked = False + + async def close(self) -> None: + self.blocked = True + self._pending_samples = 0 + if self._worker: + self._worker.cancel() + await asyncio.gather(self._worker, return_exceptions=True) + if self._stream is not None: + try: + async with self._io_lock: + await asyncio.to_thread(self._close_stream) + except Exception: + pass + + def pending_ms(self) -> float: + return self._pending_samples * 1000.0 / SAMPLE_RATE_OUT + + +@dataclass(slots=True) +class RokidRuntimeConfig: + host: str = "0.0.0.0" + port: int = 18_080 + gateway: str = "localhost:8040" + gateway_tls: bool = False + harness_url: str = "ws://127.0.0.1:8021/ws/control" + harness_client_id: str = "rokid-phase-b" + skills_config: str = "" + cleanup_mode: str = "light" + chunk_ms: int = 1_000 + force_listen_count: int = 3 + max_new_speak_tokens_per_chunk: int = 20 + length_penalty: float = 1.1 + max_slice_nums: int = 1 + audio_queue_packets: int = 96 + input_gain: float = 12.0 + image_rotate_cw: int = 270 + image_jpeg_quality: int = 95 + image_resend_s: float = 0.5 + image_max_age_s: float = 30.0 + prepare_timeout_s: float = 120.0 + close_timeout_s: float = 2.0 + reconnect_s: float = 1.5 + play_audio: bool = True + session_ready_chime: bool = True + session_ready_chime_volume: float = 0.32 + session_ready_chime_feedback_guard_s: float = 0.65 + playback_echo_tail_s: float = 0.80 + + +@dataclass(slots=True) +class SessionSpec: + generation: int + skill_id: str + slots: dict[str, Any] + system_prompt: str + + +class SessionTelemetry(Protocol): + connected: bool + + async def send(self, payload: dict[str, Any]) -> None: ... + + +class GatewayDuplexSession: + """One generation-fenced connection to the existing MiniCPM Gateway.""" + + def __init__( + self, + config: RokidRuntimeConfig, + spec: SessionSpec, + audio_queue: DropOldestAudioQueue, + latest_frame: LatestFrame, + gate: OutputGate, + on_result: Callable[["GatewayDuplexSession", dict[str, Any]], Awaitable[None]], + ): + self.config = config + self.spec = spec + self.audio_queue = audio_queue + self.latest_frame = latest_frame + self.gate = gate + self.on_result = on_result + self.session_id = ( + f"omni_rokid_pb_g{spec.generation}_{int(time.time())}_{uuid.uuid4().hex[:6]}" + ) + self.status = "created" + self.last_error = "" + self.ws: aiohttp.ClientWebSocketResponse | None = None + self._task: asyncio.Task[None] | None = None + self._ready: asyncio.Future[None] | None = None + self._stopped = asyncio.Event() + self._closing = False + self._send_lock = asyncio.Lock() + + async def _send_json(self, payload: dict[str, Any]) -> None: + ws = self.ws + if ws is None or ws.closed: + raise RuntimeError("gateway session is not connected") + async with self._send_lock: + await ws.send_json(payload) + + def _ssl_context(self) -> ssl.SSLContext | None: + if not self.config.gateway_tls: + return None + context = ssl.create_default_context() + context.check_hostname = False + context.verify_mode = ssl.CERT_NONE + return context + + async def start(self) -> None: + loop = asyncio.get_running_loop() + self._ready = loop.create_future() + self._ready.add_done_callback( + lambda future: None if future.cancelled() else future.exception() + ) + self._task = asyncio.create_task(self._run()) + await asyncio.wait_for( + asyncio.shield(self._ready), timeout=self.config.prepare_timeout_s + ) + + async def _run(self) -> None: + scheme = "wss" if self.config.gateway_tls else "ws" + url = f"{scheme}://{self.config.gateway}/ws/duplex/{self.session_id}" + self.status = "connecting" + LOG.info("[GW] connecting generation=%d %s", self.spec.generation, url) + try: + async with aiohttp.ClientSession() as client: + async with client.ws_connect( + url, + heartbeat=30, + max_msg_size=0, + ssl=self._ssl_context(), + ) as ws: + self.ws = ws + await self._prepare(ws) + self.status = "running" + if self._ready and not self._ready.done(): + self._ready.set_result(None) + send_task = asyncio.create_task(self._send_loop(ws)) + receive_task = asyncio.create_task(self._receive_loop(ws)) + done, pending = await asyncio.wait( + (send_task, receive_task), + return_when=asyncio.FIRST_COMPLETED, + ) + for task in pending: + task.cancel() + await asyncio.gather(*done, *pending, return_exceptions=True) + except asyncio.CancelledError: + raise + except Exception as exc: + self.status = "error" + self.last_error = str(exc) + if self._ready and not self._ready.done(): + self._ready.set_exception(exc) + if not self._closing: + LOG.warning("[GW] session %s failed: %s", self.session_id, exc) + finally: + self.ws = None + self._stopped.set() + if self.status != "error": + self.status = "stopped" + + async def _prepare(self, ws: aiohttp.ClientWebSocketResponse) -> None: + self.status = "queued" + while True: + message = await ws.receive() + if message.type != aiohttp.WSMsgType.TEXT: + if message.type in (aiohttp.WSMsgType.CLOSED, aiohttp.WSMsgType.ERROR): + raise RuntimeError("gateway closed while queued") + continue + payload = json.loads(message.data) + message_type = payload.get("type") + if message_type == "queue_done": + break + if message_type == "error": + raise RuntimeError(payload.get("error") or "gateway queue error") + if message_type in {"queued", "queue_update"}: + LOG.info( + "[GW] queue position=%s eta=%s", + payload.get("position"), + payload.get("estimated_wait_s"), + ) + + self.status = "preparing" + await self._send_json( + { + "type": "prepare", + "system_prompt": self.spec.system_prompt, + "config": { + "force_listen_count": self.config.force_listen_count, + "chunk_ms": self.config.chunk_ms, + "generate_audio": True, + "max_new_speak_tokens_per_chunk": ( + self.config.max_new_speak_tokens_per_chunk + ), + "length_penalty": self.config.length_penalty, + }, + "max_slice_nums": self.config.max_slice_nums, + "deferred_finalize": True, + } + ) + while True: + message = await ws.receive() + if message.type != aiohttp.WSMsgType.TEXT: + if message.type in (aiohttp.WSMsgType.CLOSED, aiohttp.WSMsgType.ERROR): + raise RuntimeError("gateway closed while preparing") + continue + payload = json.loads(message.data) + if payload.get("type") == "prepared": + LOG.info( + "[GW] prepared session=%s generation=%d skill=%s", + self.session_id, + self.spec.generation, + self.spec.skill_id, + ) + return + if payload.get("type") == "error": + raise RuntimeError(payload.get("error") or "gateway prepare error") + + async def _next_audio_chunk(self, carry: bytearray) -> np.ndarray | None: + samples = int(SAMPLE_RATE_IN * self.config.chunk_ms / 1000) + target_bytes = samples * PCM16_WIDTH + output = bytearray() + if carry: + take = min(target_bytes, len(carry)) + output.extend(carry[:take]) + del carry[:take] + deadline = time.monotonic() + max(0.1, self.config.chunk_ms / 1000 * 1.5) + while len(output) < target_bytes: + remaining = deadline - time.monotonic() + if remaining <= 0: + break + raw = await self.audio_queue.get(remaining) + if raw is None: + break + needed = target_bytes - len(output) + output.extend(raw[:needed]) + if len(raw) > needed: + carry.extend(raw[needed:]) + if not output: + return None + if len(output) < target_bytes: + output.extend(b"\x00" * (target_bytes - len(output))) + return pcm16le_to_float32(bytes(output)) + + async def _send_loop(self, ws: aiohttp.ClientWebSocketResponse) -> None: + carry = bytearray() + last_frame_sequence = -1 + last_frame_sent = 0.0 + while not self._closing: + audio = await self._next_audio_chunk(carry) + if audio is None: + continue + payload: dict[str, Any] = { + "type": "audio_chunk", + "audio_base64": float32_to_base64(audio), + } + if self.gate.speech_hold_active: + payload["force_listen"] = True + frame = self.latest_frame + frame_age_ms = now_ms() - frame.timestamp_ms + frame_due = time.monotonic() - last_frame_sent >= self.config.image_resend_s + if ( + frame.jpeg + and frame_age_ms <= self.config.image_max_age_s * 1000 + and (frame.sequence != last_frame_sequence or frame_due) + ): + payload["frame_base64_list"] = [ + base64.b64encode(frame.jpeg).decode("ascii") + ] + last_frame_sequence = frame.sequence + last_frame_sent = time.monotonic() + # 璇婃柇锛氱湡姝i佽繘 gateway 鐨勬瘡涓 chunk銆俛udio=xxpps 閭d釜缁熻鐨勬槸 + # "鎺ㄨ繘 audio_queue 鐨勫寘鏁"锛岀湅涓嶅嚭杩欓噷鏈夋病鏈夋寜 1Hz 姝e父鍙戝嚭鍘汇 + self._sent_chunks = getattr(self, "_sent_chunks", 0) + 1 + _lvl = float(np.abs(audio).mean()) if audio is not None else 0.0 + LOG.debug("[GW-SEND] #%d audio=%.2fs lvl=%.4f frame=%s hold=%s", + self._sent_chunks, + (len(audio) / SAMPLE_RATE_IN) if audio is not None else 0.0, + _lvl, + "frame_base64_list" in payload, + self.gate.speech_hold_active) + await self._send_json(payload) + + async def inject_task(self, text: str) -> bool: + """Send one text+latest-frame task to a newly prepared Skill Session.""" + text = text.strip() + if not text: + return False + payload: dict[str, Any] = { + "type": "audio_chunk", + "inject_text": text, + "max_slice_nums": self.config.max_slice_nums, + } + frame = self.latest_frame + frame_age_ms = now_ms() - frame.timestamp_ms + has_frame = bool( + frame.jpeg + and frame_age_ms <= self.config.image_max_age_s * 1000 + ) + if has_frame and frame.jpeg is not None: + payload["frame_base64_list"] = [ + base64.b64encode(frame.jpeg).decode("ascii") + ] + await self._send_json(payload) + LOG.info( + "[TASK] injected generation=%d skill=%s frame=%s text=%r", + self.spec.generation, + self.spec.skill_id, + has_frame, + text, + ) + return has_frame + + async def _receive_loop(self, ws: aiohttp.ClientWebSocketResponse) -> None: + async for message in ws: + if message.type == aiohttp.WSMsgType.TEXT: + payload = json.loads(message.data) + message_type = payload.get("type") + if message_type in {"result", "audio_only"}: + await self.on_result(self, payload) + elif message_type == "hint_audio": + # 寮哄埗鎺柦蹇垫彁绀猴細鐙珛鎾斁锛屼笉杩 on_result 鐘舵佹満 + # 锛堜笉璁 model_turn / 涓嶅彂 model.state / 涓嶈Е鍙 EchoGuard锛夛紝 + # 閬垮厤鍜 duplex 涓诲璇濇姠璇濄佺籂缂犮 + hint_b64 = str(payload.get("audio_data") or "") + if hint_b64: + try: + pcm = np.frombuffer(base64.b64decode(hint_b64), dtype=np.float32) + # 蹇垫彁绀哄墠鍏堣В闄ゅ彲鑳界殑 block锛坰top 鏃 block_and_flush 杩囷級锛 + # 鍚﹀垯 enqueue 浼氳涓€ + await self.speaker.resume() + await self.speaker.enqueue(pcm, self.spec.generation) + LOG.info("[HINT] 蹇垫彁绀烘挱鏀: %r (%d samples)", + payload.get("text"), int(pcm.size)) + except Exception as exc: + LOG.warning("[HINT] 蹇垫彁绀烘挱鏀惧け璐: %s", exc) + elif message_type == "stopped": + return + elif message_type in {"timeout", "error"}: + raise RuntimeError( + payload.get("error") or payload.get("reason") or message_type + ) + elif message.type in (aiohttp.WSMsgType.CLOSED, aiohttp.WSMsgType.ERROR): + return + + async def stop(self, cleanup_mode: str) -> None: + self._closing = True + if self.ws is not None and not self.ws.closed: + try: + await self._send_json({"type": "stop", "cleanup_mode": cleanup_mode}) + except Exception: + pass + try: + await asyncio.wait_for(self._stopped.wait(), timeout=self.config.close_timeout_s) + except asyncio.TimeoutError: + pass + if self.ws is not None and not self.ws.closed: + await self.ws.close() + if self._task and not self._task.done(): + self._task.cancel() + await asyncio.gather(self._task, return_exceptions=True) + + +SessionFactory = Callable[ + [ + RokidRuntimeConfig, + SessionSpec, + DropOldestAudioQueue, + LatestFrame, + OutputGate, + Callable[[GatewayDuplexSession, dict[str, Any]], Awaitable[None]], + ], + GatewayDuplexSession, +] + + +class GatewaySessionManager: + """Python counterpart of the frozen BrowserSessionAdapter contract.""" + + def __init__( + self, + config: RokidRuntimeConfig, + registry: SkillRegistry, + audio_queue: DropOldestAudioQueue, + latest_frame: LatestFrame, + speaker: SpeakerSink, + session_factory: SessionFactory = GatewayDuplexSession, + ): + self.config = config + self.registry = registry + self.audio_queue = audio_queue + self.latest_frame = latest_frame + self.speaker = speaker + self.session_factory = session_factory + self.gate = OutputGate(current_skill=registry.default_skill) + self.harness: SessionTelemetry | None = None + self.active: GatewayDuplexSession | None = None + self._restart_lock = asyncio.Lock() + self._playback_epoch = 0 + self._playback_release_task: asyncio.Task[None] | None = None + + async def send_telemetry(self, payload: dict[str, Any]) -> None: + if self.harness is not None: + await self.harness.send(payload) + + async def _set_playback_active(self, active: bool, source: str) -> None: + await self.send_telemetry( + { + "type": "playback.state", + "active": active, + "source": source, + "generation": self.gate.generation, + "pending_ms": round(self.speaker.pending_ms(), 1), + } + ) + + async def _mark_playback_started(self, source: str) -> None: + self._playback_epoch += 1 + epoch = self._playback_epoch + if self._playback_release_task is not None: + self._playback_release_task.cancel() + await self._set_playback_active(True, source) + + async def release_after_drain() -> None: + try: + while self.speaker.pending_ms() > 0: + await asyncio.sleep( + max(0.02, min(0.25, self.speaker.pending_ms() / 1000.0)) + ) + await asyncio.sleep(max(0.0, self.config.playback_echo_tail_s)) + if epoch == self._playback_epoch: + await self._set_playback_active(False, source) + except asyncio.CancelledError: + raise + + self._playback_release_task = asyncio.create_task(release_after_drain()) + + async def _clear_playback_state(self, source: str) -> None: + self._playback_epoch += 1 + if self._playback_release_task is not None: + self._playback_release_task.cancel() + await asyncio.gather(self._playback_release_task, return_exceptions=True) + self._playback_release_task = None + await self._set_playback_active(False, source) + + def _make_spec( + self, + skill_id: str, + slots: dict[str, Any], + system_prompt: str | None = None, + ) -> SessionSpec: + rendered = self.registry.render(skill_id, slots) + return SessionSpec( + generation=self.gate.generation, + skill_id=skill_id, + slots=dict(slots), + system_prompt=system_prompt or rendered.text, + ) + + async def start_initial(self) -> None: + async with self._restart_lock: + if self.active is not None: + return + spec = self._make_spec(self.registry.default_skill, {}) + session = self.session_factory( + self.config, + spec, + self.audio_queue, + self.latest_frame, + self.gate, + self.handle_result, + ) + self.active = session + try: + await session.start() + except Exception: + self.active = None + raise + await self._emit_bound() + + async def _emit_bound(self) -> None: + await self.send_telemetry( + { + "type": "session.state", + "phase": "bound", + "generation": self.gate.generation, + "skill_id": self.gate.current_skill, + "slots": dict(self.gate.current_slots or {}), + "session_id": self.active.session_id if self.active else None, + } + ) + + async def emit_recovery_sync(self) -> None: + if self.active is None: + return + await self.send_telemetry( + { + "type": "session.state", + "phase": "restart_complete", + "generation": self.gate.generation, + "skill_id": self.gate.current_skill, + "slots": dict(self.gate.current_slots or {}), + "new_session_id": self.active.session_id, + "cleanup_mode": "reconnect_sync", + } + ) + await self._emit_bound() + + async def handle_control(self, event: dict[str, Any]) -> dict[str, Any]: + if not event.get("accepted"): + return {"ok": False, "ignored": True} + intent = str(event.get("intent") or "") + if intent == "stop_speech": + return await self.stop_speech(event) + if intent == "resume_speech": + return await self.resume_speech(event) + if intent in { + "reset_session", + "activate_skill", + "cancel_skill", + "return_to_chat", + }: + return await self.restart(event) + return {"ok": False, "ignored": True} + + def _ack_base(self, event: dict[str, Any]) -> dict[str, Any]: + return { + "type": "control.ack", + "event_id": event.get("event_id"), + "intent": event.get("intent"), + "generation": self.gate.generation, + "dropped_old_text": self.gate.dropped_old_text, + "dropped_old_audio": self.gate.dropped_old_audio, + } + + async def stop_speech( + self, event: dict[str, Any], *, emit_ack: bool = True + ) -> dict[str, Any]: + self.gate.stop() + await self.speaker.block_and_flush() + await self._clear_playback_state("stop") + ack = {**self._ack_base(event), "ok": True} + if emit_ack: + await self.send_telemetry(ack) + LOG.info("[CONTROL] STOP event=%s", event.get("event_id")) + return ack + + async def resume_speech(self, event: dict[str, Any]) -> dict[str, Any]: + self.gate.resume() + await self.speaker.resume() + ack = {**self._ack_base(event), "ok": True} + await self.send_telemetry(ack) + LOG.info("[CONTROL] RESUME event=%s", event.get("event_id")) + return ack + + async def restart(self, event: dict[str, Any]) -> dict[str, Any]: + async with self._restart_lock: + requested_skill = str(event.get("skill_id") or self.registry.default_skill) + requested_slots = dict(event.get("slots") or {}) + if ( + str(event.get("intent")) != "reset_session" + and requested_skill == self.gate.current_skill + and same_slots(requested_slots, dict(self.gate.current_slots or {})) + ): + ack = {**self._ack_base(event), "ok": True, "no_restart": True} + await self.send_telemetry(ack) + return ack + + started = time.monotonic() + await self.stop_speech(event, emit_ack=False) + self.gate.restart_in_progress = True + old_session = self.active + old_session_id = old_session.session_id if old_session else None + self.gate.generation += 1 + await self.send_telemetry( + { + "type": "session.state", + "phase": "restart_started", + "generation": self.gate.generation, + "event_id": event.get("event_id"), + "old_session_id": old_session_id, + } + ) + self.active = None + try: + if old_session: + await old_session.stop(self.config.cleanup_mode) + self.audio_queue.clear() + spec = self._make_spec( + requested_skill, + requested_slots, + str(event.get("system_prompt") or "") or None, + ) + session = self.session_factory( + self.config, + spec, + self.audio_queue, + self.latest_frame, + self.gate, + self.handle_result, + ) + self.active = session + await session.start() + self.gate.current_skill = requested_skill + self.gate.current_slots = requested_slots + self.gate.replacement_ready() + await self.speaker.resume() + await self._emit_bound() + latency_ms = (time.monotonic() - started) * 1000.0 + state = { + "type": "session.state", + "phase": "restart_complete", + "generation": self.gate.generation, + "skill_id": requested_skill, + "slots": requested_slots, + "old_session_id": old_session_id, + "new_session_id": session.session_id, + "cleanup_mode": self.config.cleanup_mode, + } + await self.send_telemetry(state) + if self.config.play_audio and self.config.session_ready_chime: + cue = make_session_ready_chime( + amplitude=self.config.session_ready_chime_volume + ) + self.gate.local_cue_mute_until_mono = max( + self.gate.local_cue_mute_until_mono, + time.monotonic() + + cue.size / SAMPLE_RATE_OUT + + self.config.session_ready_chime_feedback_guard_s, + ) + await self.speaker.enqueue(cue, self.gate.generation) + await self._mark_playback_started("session_ready_chime") + LOG.info( + "[CUE] session ready generation=%d skill=%s", + self.gate.generation, + requested_skill, + ) + trigger = self.registry.task_trigger(requested_skill, requested_slots) + trigger_frame = False + trigger_error = "" + if trigger: + try: + trigger_frame = await session.inject_task(trigger) + await self.send_telemetry( + { + "type": "session.state", + "phase": "task_trigger_sent", + "generation": self.gate.generation, + "skill_id": requested_skill, + "slots": requested_slots, + "session_id": session.session_id, + "has_frame": trigger_frame, + } + ) + except Exception as exc: + trigger_error = str(exc) + LOG.warning( + "[TASK] one-shot trigger failed skill=%s: %s", + requested_skill, + exc, + ) + ack = { + **self._ack_base(event), + "ok": True, + "restart_latency_ms": latency_ms, + "skill_id": requested_skill, + "slots": requested_slots, + "old_session_id": old_session_id, + "new_session_id": session.session_id, + "cleanup_mode": self.config.cleanup_mode, + "task_trigger_sent": bool(trigger and not trigger_error), + "task_trigger_has_frame": trigger_frame, + "task_trigger_error": trigger_error, + } + await self.send_telemetry(ack) + LOG.info( + "[CONTROL] %s ready generation=%d session=%s latency_ms=%.1f", + requested_skill, + self.gate.generation, + session.session_id, + latency_ms, + ) + return ack + except Exception as exc: + if self.active: + await self.active.stop(self.config.cleanup_mode) + self.active = None + self.gate.restart_in_progress = False + ack = {**self._ack_base(event), "ok": False, "error": str(exc)} + await self.send_telemetry(ack) + LOG.exception("[CONTROL] restart failed") + return ack + + async def handle_result( + self, session: GatewayDuplexSession, result: dict[str, Any] + ) -> None: + is_listen = bool(result.get("is_listen")) + text = str(result.get("text") or "") + audio_b64 = str(result.get("audio_data") or "") + stale = ( + session is not self.active + or session.spec.generation != self.gate.generation + ) + if stale or (self.gate.drop_output_until_listen and not is_listen): + if text: + self.gate.dropped_old_text += 1 + if audio_b64: + self.gate.dropped_old_audio += 1 + # 璇婃柇锛氳繖鏉¤矾寰勫師鏈潤榛 return锛屾ā鍨嬬殑杈撳嚭琚暣涓涪鎺夊嵈涓嶇暀鐥曡抗 + #锛堣〃鐜板氨鏄 ai(calls=0) + [MODEL] 鍙湁瀛ら浂闆朵竴鏉★級銆傛墦鍑轰涪寮冨師鍥犮 + self._drop_n = getattr(self, "_drop_n", 0) + 1 + if self._drop_n <= 3 or self._drop_n % 100 == 0: + LOG.warning( + "[GW-DROP] #%d stale=%s (active=%s gen_sess=%s gen_gate=%s) " + "drop_until_listen=%s is_listen=%s text=%r audio=%dB", + self._drop_n, stale, + session is self.active, + session.spec.generation, self.gate.generation, + self.gate.drop_output_until_listen, is_listen, + text[:20], len(audio_b64)) + return + if is_listen and self.gate.drop_output_until_listen: + if not self.gate.speech_hold_active: + self.gate.drop_output_until_listen = False + await self.speaker.resume() + audio_samples = 0 + if audio_b64 and not is_listen: + try: + pcm = np.frombuffer(base64.b64decode(audio_b64), dtype=np.float32) + audio_samples = int(pcm.size) + await self.speaker.enqueue(pcm, session.spec.generation) + if self.config.play_audio: + await self._mark_playback_started("model") + except Exception as exc: + LOG.warning("model audio decode failed: %s", exc) + await self.send_telemetry( + { + "type": "model.state", + "state": "listen" if is_listen else "speak", + "text": text, + "generation": session.spec.generation, + "session_id": session.session_id, + "skill_id": session.spec.skill_id, + "slots": dict(session.spec.slots), + # llama.cpp-omni marks each decode slice as end_of_turn. The + # actual duplex turn boundary is the later __IS_LISTEN__ + # result, so keep the raw marker for diagnostics but aggregate + # model text until listening resumes. + "decode_end": bool(result.get("end_of_turn")), + "end_of_turn": is_listen, + "message_type": str(result.get("type") or "result"), + "audio_samples": audio_samples, + "audio_ms": round( + audio_samples * 1000.0 / SAMPLE_RATE_OUT, 1 + ), + } + ) + if text: + LOG.info("[MODEL] listen=%s text=%s", is_listen, text) + + async def close(self) -> None: + await self._clear_playback_state("close") + if self.active: + await self.active.stop(self.config.cleanup_mode) + self.active = None + + def health(self) -> dict[str, Any]: + return { + "generation": self.gate.generation, + "current_skill": self.gate.current_skill, + "current_slots": dict(self.gate.current_slots or {}), + "speech_hold_active": self.gate.speech_hold_active, + "restart_in_progress": self.gate.restart_in_progress, + "dropped_old_text": self.gate.dropped_old_text, + "dropped_old_audio": self.gate.dropped_old_audio, + "session_id": self.active.session_id if self.active else None, + "gateway_status": self.active.status if self.active else "disconnected", + "gateway_error": self.active.last_error if self.active else "", + } + + +class HarnessClient: + def __init__( + self, + url: str, + client_id: str, + manager: GatewaySessionManager, + reconnect_s: float, + ): + separator = "&" if "?" in url else "?" + self.url = f"{url}{separator}client_id={client_id}" + self.manager = manager + self.reconnect_s = reconnect_s + self.connected = False + self.ws: aiohttp.ClientWebSocketResponse | None = None + self._stop = asyncio.Event() + self._send_lock = asyncio.Lock() + self._control_tasks: set[asyncio.Task[Any]] = set() + + # 澶栭儴鍙寕鐨勫彧璇昏瀵熻咃細鏀跺埌鏈涓婇潰鍒嗘敮澶勭悊鐨勬秷鎭椂璋冪敤锛堝悓姝ャ佸紓甯稿悶鎺夛級銆 + # 涓嶆敼鏋勯犵鍚嶏紝璧嬪煎嵆鍙細client.on_message = fn + on_message = None + + async def run(self) -> None: + while not self._stop.is_set(): + try: + async with aiohttp.ClientSession() as client: + async with client.ws_connect( + self.url, heartbeat=30, max_msg_size=0 + ) as ws: + self.ws = ws + self.connected = True + LOG.info("Harness connected: %s", self.url) + async for message in ws: + if message.type != aiohttp.WSMsgType.TEXT: + if message.type in ( + aiohttp.WSMsgType.CLOSED, + aiohttp.WSMsgType.ERROR, + ): + break + continue + payload = json.loads(message.data) + message_type = payload.get("type") + if message_type == "harness.ready": + await self.manager.emit_recovery_sync() + elif message_type == "control.intent": + task = asyncio.create_task( + self.manager.handle_control(payload) + ) + self._control_tasks.add(task) + task.add_done_callback(self._control_tasks.discard) + # 鍙夋梺璺細鎶婂叾浣欐秷鎭紙濡 asr.transcript锛変氦缁欏閮ㄨ瀵熻呫 + # 榛樿 None锛岃涓轰笌鍘熸潵瀹屽叏涓鑷淬8021 鐨 asr.transcript 鍙姇閫掔粰 + # "閫侀煶棰戣繘鏉ョ殑閭d釜 client" 鐨 outbound 闃熷垪锛堟瘡涓 client_id 鏈 + # 鐙珛 runtime锛夛紝鍙﹀紑涓鏉¤繛鎺ユ槸鏀朵笉鍒扮殑锛屽彧鑳藉湪杩欓噷鎴 + elif self.on_message is not None: + try: + self.on_message(payload) + except Exception: + pass + except asyncio.CancelledError: + raise + except Exception as exc: + if not self._stop.is_set(): + LOG.warning("Harness connection failed: %s", exc) + finally: + self.connected = False + self.ws = None + if not self._stop.is_set(): + await asyncio.sleep(self.reconnect_s) + + async def send(self, payload: dict[str, Any]) -> None: + async with self._send_lock: + if self.ws is not None and not self.ws.closed: + await self.ws.send_json(payload) + + async def send_audio(self, samples: np.ndarray) -> None: + if samples.size == 0: + return + await self.send( + { + "type": "audio.mirror", + "started_at_ms": now_ms() - samples.size * 1000.0 / SAMPLE_RATE_IN, + "sample_rate": SAMPLE_RATE_IN, + "audio_b64": float32_to_base64(samples), + } + ) + + async def send_frame(self, jpeg: bytes, sequence: int, timestamp_ms: float) -> None: + await self.send( + { + "type": "frame.shadow", + "frame_id": f"rokid_{sequence}_{int(timestamp_ms)}", + "timestamp_ms": timestamp_ms, + "jpeg_b64": base64.b64encode(jpeg).decode("ascii"), + } + ) + + async def close(self) -> None: + self._stop.set() + if self.ws is not None and not self.ws.closed: + await self.ws.close() + for task in tuple(self._control_tasks): + task.cancel() + await asyncio.gather(*self._control_tasks, return_exceptions=True) + + +@dataclass(slots=True) +class RuntimeStats: + started_mono: float = field(default_factory=time.monotonic) + audio_packets: int = 0 + audio_bytes: int = 0 + audio_clients: int = 0 + image_count: int = 0 + image_bytes: int = 0 + last_audio_ms: float = 0.0 + audio_rms: float = 0.0 + audio_peak: float = 0.0 + non_silent_audio_packets: int = 0 + + +class PhaseBRokidRuntime: + """Rokid sensor ingress + frozen Harness control + Gateway session adapter.""" + + def __init__( + self, + config: RokidRuntimeConfig, + *, + speaker: SpeakerSink | None = None, + session_factory: SessionFactory = GatewayDuplexSession, + ): + self.config = config + self.registry = SkillRegistry(config.skills_config) + self.audio_queue = DropOldestAudioQueue(config.audio_queue_packets) + self.audio_mirror = AudioMirrorChunker() + self.latest_frame = LatestFrame() + self.stats = RuntimeStats(started_mono=time.monotonic()) + self.speaker = speaker or (PCSpeaker() if config.play_audio else NullSpeaker()) + self.manager = GatewaySessionManager( + config, + self.registry, + self.audio_queue, + self.latest_frame, + self.speaker, + session_factory=session_factory, + ) + self.harness = HarnessClient( + config.harness_url, + config.harness_client_id, + self.manager, + config.reconnect_s, + ) + self.manager.harness = self.harness + self._tasks: list[asyncio.Task[Any]] = [] + self._close_lock = asyncio.Lock() + self._closed = False + self._no_input_warning_emitted = False + + def health(self) -> dict[str, Any]: + return { + "ok": True, + "phase": "B", + "device": "rokid", + "uptime_s": round(time.monotonic() - self.stats.started_mono, 3), + "harness_connected": self.harness.connected, + "audio_packets_in": self.stats.audio_packets, + "audio_bytes_in": self.stats.audio_bytes, + "audio_clients": self.stats.audio_clients, + "latest_audio_age_ms": ( + round(now_ms() - self.stats.last_audio_ms, 1) + if self.stats.last_audio_ms + else None + ), + "audio_queue_size": self.audio_queue.size, + "dropped_audio_packets": self.audio_queue.dropped_packets, + "audio_rms": round(self.stats.audio_rms, 6), + "audio_peak": round(self.stats.audio_peak, 6), + "input_gain": self.config.input_gain, + "effective_audio_rms": round( + min(1.0, self.stats.audio_rms * self.config.input_gain), 6 + ), + "non_silent_audio_packets": self.stats.non_silent_audio_packets, + "image_count": self.stats.image_count, + "device_input_ready": bool( + self.stats.audio_packets or self.stats.image_count + ), + "latest_image_age_ms": ( + round(now_ms() - self.latest_frame.timestamp_ms, 1) + if self.latest_frame.timestamp_ms + else None + ), + **self.manager.health(), + } + + async def _initial_session_loop(self) -> None: + while True: + try: + await self.manager.start_initial() + return + except asyncio.CancelledError: + raise + except Exception as exc: + LOG.warning("initial Gateway session failed: %s", exc) + await asyncio.sleep(self.config.reconnect_s) + + async def _stats_loop(self) -> None: + previous_audio = 0 + previous_image = 0 + while True: + await asyncio.sleep(5.0) + if ( + not self._no_input_warning_emitted + and time.monotonic() - self.stats.started_mono >= 10.0 + and self.stats.audio_packets == 0 + and self.stats.image_count == 0 + ): + self._no_input_warning_emitted = True + LOG.warning( + "[ROKID][NO_INPUT] no PCM/JPEG has reached this process. " + "After the PC runtime is listening, restart OpenGlass " + "Sensor Mode on the glasses (STOP -> RUN) and verify that " + "the APK targets the PC WLAN address on port %d", + self.config.port, + ) + LOG.info( + "[STATS] audio=%.1fpps images=%d(+%d) audio_clients=%d " + "queue=%d drops=%d rms=%.4f peak=%.4f harness=%s gateway=%s " + "skill=%s generation=%d", + (self.stats.audio_packets - previous_audio) / 5.0, + self.stats.image_count, + self.stats.image_count - previous_image, + self.stats.audio_clients, + self.audio_queue.size, + self.audio_queue.dropped_packets, + self.stats.audio_rms, + self.stats.audio_peak, + self.harness.connected, + self.manager.health()["gateway_status"], + self.manager.gate.current_skill, + self.manager.gate.generation, + ) + previous_audio = self.stats.audio_packets + previous_image = self.stats.image_count + + async def start(self) -> None: + await self.speaker.start() + self._tasks = [ + asyncio.create_task(self.harness.run()), + asyncio.create_task(self._initial_session_loop()), + asyncio.create_task(self._stats_loop()), + ] + + async def close(self) -> None: + async with self._close_lock: + if self._closed: + return + LOG.info("Rokid runtime shutdown started") + + async def bounded(label: str, awaitable: Awaitable[None]) -> None: + try: + await asyncio.wait_for( + awaitable, + timeout=max(0.5, self.config.close_timeout_s), + ) + except asyncio.TimeoutError: + LOG.warning( + "Rokid shutdown step timed out: %s (continuing)", label + ) + except Exception as exc: + LOG.warning( + "Rokid shutdown step failed: %s: %s", label, exc + ) + + await bounded("harness", self.harness.close()) + for task in self._tasks: + task.cancel() + if self._tasks: + await bounded( + "background_tasks", + asyncio.gather(*self._tasks, return_exceptions=True), + ) + self._tasks.clear() + await bounded("gateway_session", self.manager.close()) + await bounded("pc_speaker", self.speaker.close()) + self.audio_queue.clear() + self.audio_mirror.clear() + self._closed = True + LOG.info("Rokid runtime shutdown complete; port may be reused") + + def create_app(self) -> web.Application: + app = web.Application(client_max_size=8 * 1024 * 1024) + app["runtime"] = self + + async def health(_request: web.Request) -> web.Response: + return web.json_response(self.health()) + + async def capture(_request: web.Request) -> web.Response: + if not self.latest_frame.jpeg: + return web.Response(status=404, text="no image") + return web.Response(body=self.latest_frame.jpeg, content_type="image/jpeg") + + async def rokid_image(request: web.Request) -> web.Response: + raw = await request.read() + if not (raw.startswith(b"\xff\xd8") and raw.endswith(b"\xff\xd9")): + return web.Response(status=400, text="expected JPEG bytes") + try: + jpeg = await asyncio.to_thread( + rotate_jpeg_clockwise, + raw, + self.config.image_rotate_cw, + self.config.image_jpeg_quality, + ) + except Exception as exc: + return web.Response(status=400, text=str(exc)) + timestamp = now_ms() + self.latest_frame.set(jpeg, timestamp) + self.stats.image_count += 1 + self.stats.image_bytes += len(raw) + if self.stats.image_count == 1: + LOG.info("[ROKID] first JPEG received from %s", request.remote) + asyncio.create_task( + self.harness.send_frame( + jpeg, self.latest_frame.sequence, timestamp + ) + ) + return web.json_response( + {"ok": True, "image_count": self.stats.image_count, "bytes": len(raw)} + ) + + async def rokid_audio(request: web.Request) -> web.WebSocketResponse: + ws = web.WebSocketResponse(heartbeat=30, max_msg_size=0) + await ws.prepare(request) + self.stats.audio_clients += 1 + LOG.info("[ROKID] audio connected from %s", request.remote) + try: + async for message in ws: + if message.type == aiohttp.WSMsgType.BINARY: + raw = bytes(message.data) + if len(raw) < PCM16_WIDTH: + continue + if len(raw) % PCM16_WIDTH: + raw = raw[:-1] + self.stats.audio_packets += 1 + self.stats.audio_bytes += len(raw) + self.stats.last_audio_ms = now_ms() + if self.stats.audio_packets == 1: + LOG.info("[ROKID] first PCM packet received") + samples = pcm16le_to_float32(raw) + packet_rms = float(np.sqrt(np.mean(np.square(samples)))) + packet_peak = float(np.max(np.abs(samples))) + self.stats.audio_rms = ( + 0.8 * self.stats.audio_rms + 0.2 * packet_rms + ) + self.stats.audio_peak = packet_peak + if packet_rms >= 0.01: + self.stats.non_silent_audio_packets += 1 + if time.monotonic() >= self.manager.gate.local_cue_mute_until_mono: + amplified = apply_pcm16_gain(raw, self.config.input_gain) + self.audio_queue.put_nowait(amplified) + for frame in self.audio_mirror.feed(amplified): + await self.harness.send_audio(frame) + elif message.type == aiohttp.WSMsgType.ERROR: + break + finally: + self.stats.audio_clients = max(0, self.stats.audio_clients - 1) + LOG.info("[ROKID] audio disconnected from %s", request.remote) + return ws + + async def on_startup(_app: web.Application) -> None: + await self.start() + + async def on_cleanup(_app: web.Application) -> None: + await self.close() + + app.router.add_get("/", health) + app.router.add_get("/health", health) + app.router.add_get("/capture", capture) + app.router.add_post("/rokid/image", rokid_image) + app.router.add_get("/rokid/audio", rokid_audio) + app.on_startup.append(on_startup) + app.on_cleanup.append(on_cleanup) + return app + + +def default_skills_config() -> str: + return str( + Path(__file__).resolve().parents[1] / "config" / "skills.example.yaml" + ) + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Rokid Phase B adapter: sensors + Harness + MiniCPM Gateway" + ) + parser.add_argument("--host", default="0.0.0.0") + parser.add_argument("--port", type=int, default=18_080) + parser.add_argument("--gateway", default="localhost:8040") + parser.add_argument("--gateway-tls", action="store_true", default=False) + parser.add_argument( + "--no-gateway-tls", dest="gateway_tls", action="store_false" + ) + parser.add_argument( + "--harness-url", default="ws://127.0.0.1:8021/ws/control" + ) + parser.add_argument("--client-id", default="rokid-phase-b") + parser.add_argument("--skills-config", default=default_skills_config()) + parser.add_argument("--cleanup-mode", choices=("light", "full"), default="light") + parser.add_argument("--image-rotate-cw", type=int, default=270) + parser.add_argument("--image-jpeg-quality", type=int, default=95) + parser.add_argument("--audio-queue-packets", type=int, default=96) + parser.add_argument( + "--input-gain", + type=float, + default=12.0, + help="Rokid PCM gain before Harness/Gateway; clipped to PCM16", + ) + parser.add_argument("--chunk-ms", type=int, default=1_000) + parser.add_argument("--force-listen-count", type=int, default=3) + parser.add_argument("--no-play", action="store_true") + parser.add_argument( + "--no-session-ready-chime", + action="store_true", + help="disable the local restart_complete notification cue", + ) + parser.add_argument( + "--session-ready-chime-volume", + type=float, + default=0.32, + help="restart cue amplitude in [0, 1] (default: 0.32)", + ) + parser.add_argument( + "--playback-echo-tail-s", + type=float, + default=0.80, + help="keep EchoGuard active after the PC speaker queue drains", + ) + parser.add_argument("--log-level", default="INFO") + return parser.parse_args() + + +def main() -> None: + args = parse_args() + if not 0.0 <= args.session_ready_chime_volume <= 1.0: + raise SystemExit("--session-ready-chime-volume must be between 0 and 1") + if args.playback_echo_tail_s < 0.0: + raise SystemExit("--playback-echo-tail-s must be non-negative") + logging.basicConfig( + level=getattr(logging, args.log_level.upper(), logging.INFO), + format="%(asctime)s %(levelname)s %(name)s: %(message)s", + force=True, + ) + config = RokidRuntimeConfig( + host=args.host, + port=args.port, + gateway=args.gateway, + gateway_tls=args.gateway_tls, + harness_url=args.harness_url, + harness_client_id=args.client_id, + skills_config=args.skills_config, + cleanup_mode=args.cleanup_mode, + chunk_ms=args.chunk_ms, + force_listen_count=args.force_listen_count, + audio_queue_packets=args.audio_queue_packets, + input_gain=args.input_gain, + image_rotate_cw=args.image_rotate_cw, + image_jpeg_quality=args.image_jpeg_quality, + play_audio=not args.no_play, + session_ready_chime=not args.no_session_ready_chime, + session_ready_chime_volume=args.session_ready_chime_volume, + playback_echo_tail_s=args.playback_echo_tail_s, + ) + runtime = PhaseBRokidRuntime(config) + watchdog = CtrlCExitWatchdog() + shutdown_signals = [signal.SIGINT] + if hasattr(signal, "SIGBREAK"): + shutdown_signals.append(signal.SIGBREAK) + previous_handlers = { + signum: signal.getsignal(signum) for signum in shutdown_signals + } + + def handle_console_shutdown(signum: int, frame: Any) -> None: + watchdog.arm() + previous_handler = previous_handlers[signum] + if callable(previous_handler): + previous_handler(signum, frame) + else: + raise KeyboardInterrupt + + for signum in shutdown_signals: + signal.signal(signum, handle_console_shutdown) + LOG.info("Rokid Phase B input: http://%s:%d", config.host, config.port) + LOG.info("Harness: %s", config.harness_url) + LOG.info("Gateway: %s://%s", "wss" if config.gateway_tls else "ws", config.gateway) + try: + web.run_app( + runtime.create_app(), + host=config.host, + port=config.port, + access_log=None, + shutdown_timeout=max(0.5, config.close_timeout_s), + handler_cancellation=True, + ) + except OSError as exc: + if getattr(exc, "winerror", None) == 10048: + LOG.error( + "Port %d is already occupied by another process. A runtime " + "started with this patched version exits within %.0fs after " + "Ctrl+C.", + config.port, + CTRL_C_HARD_EXIT_S, + ) + raise SystemExit(2) from None + raise + finally: + for signum, previous_handler in previous_handlers.items(): + signal.signal(signum, previous_handler) + if watchdog.armed: + LOG.info("Graceful shutdown returned; exiting Rokid process") + + +if __name__ == "__main__": + main() diff --git a/extensions/assistive_harness/phase_b/session_recorder.py b/extensions/assistive_harness/phase_b/session_recorder.py new file mode 100644 index 0000000..ce6bcda --- /dev/null +++ b/extensions/assistive_harness/phase_b/session_recorder.py @@ -0,0 +1,90 @@ +"""Session 褰曞埗鍣紙瀵归綈 -v 鐨 SessionRecorder 鏍煎紡锛夈 + +娌跨敤 pc_vlm_v5_funnel.SessionRecorder 鐨勮惤鐩樼粨鏋勶紝璁 -o 閾捐矾褰曞嚭鐨 session +鍜 -v 瀹屽叏涓鑷达紝鍙敤 -v 鐨勫垎鏋 / rerun 宸ュ叿閾剧洿鎺ヨ銆 + + sessions// + images/q{NNN}_f{MM}.jpg 姣忚疆涓绨囧抚锛堟暣绨囧瓨锛 + queries.jsonl 姣忚涓鏉★紝-v 鍘熷瓧娈 + 婕忔枟鍒ゅ畾瀛楁 + meta.json + +鍚姩鏍囧織锛-v 鏄孉SR 鍑 query銆嶏紝-o 閲屾崲鎴愩岄摼璺惎鍔ㄣ嶁斺斾粠 start 璧锋寔缁綍锛 +姣忚疆婕忔枟鍐崇瓥 append 涓鏉°俼id 鍗曡皟閫掑锛堜竴杞 = 涓涓 qid锛夈 + +queries.jsonl 姣忚锛堜繚鐣 -v 鍘熷瓧娈 qid/text/frames/ts锛屽吋瀹癸紱杩藉姞婕忔枟瀛楁锛夛細 + { + "qid": 3, "ts": 1699..., # -v 鍘熷瓧娈 + "text": "", # -v 鍘熷瓧娈碉紙-o 鏃 ASR query 鏂囨湰锛岀暀绌/鍙悗濉級 + "frames": ["q003_f00.jpg", ...], # -v 鍘熷瓧娈碉紙鏁寸皣甯ф枃浠跺悕锛 + "reason": "severe_shake", # 婕忔枟鍒ゅ畾 + "send": false, + "hint": "鏅冨緱鍘夊锛岃鎷跨ǔ涓涓", + "best_index": 1, + "af_triggered": true, "af_ok": true, + "timings": {...} + } +""" +from __future__ import annotations + +import json +import time +from pathlib import Path +from typing import Optional + + +class SessionRecorder: + def __init__(self, sessions_root: str = "sessions", sid: Optional[str] = None): + sid = sid or time.strftime("%Y%m%d_%H%M%S") + self.dir = Path(sessions_root) / sid + self.images = self.dir / "images" + self.images.mkdir(parents=True, exist_ok=True) + self.jsonl = self.dir / "queries.jsonl" + (self.dir / "meta.json").write_text( + json.dumps({"sid": sid, "source": "esp32_runtime", "created": time.time()}, + ensure_ascii=False, indent=2), + encoding="utf-8") + self._qid = 0 + print(f"[SESSION] recording to {self.dir}") + + def record(self, decision, text: str = "") -> None: + """瀛樹竴杞紡鏂楀喅绛栵細鏁寸皣甯ц惤鐩 + queries.jsonl append 涓鏉° + + decision: FunnelDecision锛堝惈 frames / best / reason / hint / + best_index / af_triggered / af_ok / timings锛 + text: 鍙夛紝-o 鏃 ASR query 鏂囨湰锛岀暀绌猴紱濡備笂灞傛湁 ASR 缁撴灉鍙紶鍏ャ + """ + self._qid += 1 + qid = self._qid + frame_names = [] + frames = decision.frames or ([] if decision.best is None else [decision.best]) + for i, jpg in enumerate(frames): + name = f"q{qid:03d}_f{i:02d}.jpg" + try: + (self.images / name).write_bytes(jpg) + frame_names.append(name) + except Exception: + pass + + rec = { + # -v 鍘熷瓧娈碉紙鍏煎 -v 宸ュ叿閾撅級 + "qid": qid, + "ts": time.time(), + "text": text, + "frames": frame_names, + # 婕忔枟鍒ゅ畾瀛楁 + "send": bool(decision.send), + "reason": decision.reason, + "hint": decision.hint, + "best_index": decision.best_index, + "af_triggered": decision.af_triggered, + "af_ok": decision.af_ok, + "timings": decision.timings, + } + try: + with self.jsonl.open("a", encoding="utf-8") as f: + f.write(json.dumps(rec, ensure_ascii=False) + "\n") + except Exception: + pass + + def close(self) -> None: + print(f"[SESSION] recorded {self._qid} queries -> {self.dir}") diff --git a/extensions/assistive_harness/phase_b/templates/live.html b/extensions/assistive_harness/phase_b/templates/live.html new file mode 100644 index 0000000..5ab0032 --- /dev/null +++ b/extensions/assistive_harness/phase_b/templates/live.html @@ -0,0 +1,625 @@ + + + + + +AI 鐪奸暅 路 瀹炴椂婕旂ず + + + + +
+
+
+ 馃暥 +
+
AI 鐪奸暅 路 瀹炴椂婕旂ず
+
Real-time multimodal assistant for the visually impaired
+
+
+
+ + 杩炴帴涓 + + + 寰呰繛鎺 + + + 杈撳叆寮傚父 + + 馃搧 鍘嗗彶 + +
+
+ +
+
+ +
+
馃摲
+
绛夊緟鐪奸暅鐢婚潰鈥
+
+
+ LIVE 路 鐪奸暅绗竴瑙嗚 +
+
+ 馃摗 鍓嶆柟 -- m + 馃Л 鏈濆悜 --掳 +
+
+ 馃帳 +
+ -- dB +
+
+
+
+ +
+
+ 馃挰 瀵硅瘽 + 0 杞 +
+
+
寮濮嬭璇濅互涓 AI 鐪奸暅浜掑姩鈥
+
+
+
+ +
+
+ AI 闈欓粯涓 +
+
+
+ + +
+
+
+ +
鈿 鐢ㄦ埛鎵撴柇浜 AI
+ + + + + +
+
+

鍋滄骞朵繚瀛?

+

灏嗙珛鍗崇粨鏉熸湰娆′細璇,骞舵妸瑙嗛 / 闊抽 / 瀛楀箷鎵撳寘鍐欑洏銆傜‘璁ゅ悗鏃犳硶鎭㈠銆

+
+ + +
+
+
+ + + + \ No newline at end of file diff --git a/extensions/assistive_harness/phase_b/templates/replay.html b/extensions/assistive_harness/phase_b/templates/replay.html new file mode 100644 index 0000000..458d238 --- /dev/null +++ b/extensions/assistive_harness/phase_b/templates/replay.html @@ -0,0 +1,566 @@ + + + + + + AI 鐪奸暅 路 鍥炴斁 + + + + +
+
+
+ 馃幀 +
+
AI 鐪奸暅 路 浼氳瘽鍥炴斁
+
+
+
+
+ 鎺㈡祴涓 + 0 杞 + 猬 鎵鏈変細璇 + 馃敶 瀹炴椂 +
+
+ +
+
+
+
+
鍔犺浇涓
+
+
+ +
+
+ 馃挰 瀛楀箷鏃堕棿绾 + 鐐瑰嚮璺宠浆 +
+
+
鏆傛棤瀛楀箷
+
+
+
+ +
+ +
+
+
+ + + + \ No newline at end of file diff --git a/extensions/assistive_harness/phase_b/templates/replay_index.html b/extensions/assistive_harness/phase_b/templates/replay_index.html new file mode 100644 index 0000000..efcc60a --- /dev/null +++ b/extensions/assistive_harness/phase_b/templates/replay_index.html @@ -0,0 +1,321 @@ + + + + + + AI 鐪奸暅 路 鍥炶 + + + + +
+
+
+ 馃暥 +
+
AI 鐪奸暅 路 浼氳瘽鍥炶
+
Session replay & offline inspection
+
+
+
+ 鈥 鏉′細璇 + 猬 杩斿洖瀹炴椂 +
+
+ +
+
+
+ 鐐瑰嚮琛岃繘鍏ュ洖鏀 路 鎵鏈変細璇濇寜鏃堕棿鍊掑簭 +
+
+
Loading鈥
+
+
+
+
+ + + + \ No newline at end of file diff --git a/extensions/assistive_harness/phase_b/timing_probe.py b/extensions/assistive_harness/phase_b/timing_probe.py new file mode 100644 index 0000000..6d921c4 --- /dev/null +++ b/extensions/assistive_harness/phase_b/timing_probe.py @@ -0,0 +1,163 @@ +"""鏃跺簭鎺㈤拡锛堢函鏃佽矾锛屼笉鏀归摼璺涓猴級銆 + +鍙仛涓浠朵簨锛氭妸 ESP32 -o 閾捐矾閲"鍙栧浘 / 闊抽 / 鐩镐綅"鐨勬椂闂寸偣鍐欒繘涓涓 CSV锛 +渚涗簨鍚庡垎鏋愭紡鏂楄鎻掑湪鍝佽法鍑犱釜 chunk銆侀煶鍥炬庝箞鎶€ + +涓嶆帴瀵圭劍銆佷笉鎺ユ紡鏂楀垽瀹 鈥斺 杩欐槸绾椂搴忔帰閽堛 + +鍒楀悕灏介噺娌跨敤 -v 鐨 latency CSV 椋庢牸锛坓rab_ms / n_frames 绛夛級锛 +棰濆鍔 -o 鐗规湁鐨 chunk / 闊抽鐩镐綅鍒楋紝鏂逛究鍜 -v 鏁版嵁瀵规瘮銆佷篃璁╁垎鏋愬伐鍏烽摼澶嶇敤銆 + +鐢ㄦ硶锛堝湪 esp32_runtime 閲岋級锛 + from .timing_probe import TimingProbe + probe = TimingProbe(enabled=args.timing_probe, path=args.timing_csv, chunk_ms=1000) + # 鍙栧浘锛 + probe.mark_grab(seq, grab_ms, jpeg_bytes, since_last_ms) + # 闊抽锛堟瘡鍖呰皟锛屽唴閮ㄦ寜绉掕仛鍚堬級锛 + probe.mark_audio(seq, n_samples, drops, rms) + # 鍏抽棴锛歱robe.close() +""" +from __future__ import annotations + +import csv +import time +from pathlib import Path +from typing import Optional + + +class TimingProbe: + def __init__(self, enabled: bool = True, path: Optional[str] = None, + chunk_ms: int = 1000): + self.enabled = enabled + self.chunk_ms = chunk_ms + self._t0 = time.monotonic() # 杩涚▼璧风偣锛岀敤浜庣畻 rel_ms / chunk_id + self._f = None + self._w = None + + # 鈹鈹 闊抽娲诲姩/闈欓粯鐩镐綅鍒ゅ畾锛堢敤浜庣粰姣忓抚鍥炬爣 phase锛夆攢鈹 + # 绠鍗曡兘閲忛棬闄愶細rms > 闃堝肩畻"娲诲姩"锛屽惁鍒"闈欓粯"銆傞槇鍊煎彲璋冦 + self._voice_rms_gate = 0.01 + self._last_audio_active_rel_ms = -1e9 # 鏈杩戜竴娆¢煶棰戞椿鍔ㄧ殑鐩稿鏃跺埢 + self._audio_active = False + + # 鈹鈹 闊抽鎸夌鑱氬悎缂撳瓨 鈹鈹 + self._agg_bucket = -1 # 褰撳墠鑱氬悎鍒扮鍑犵 + self._agg_pkts = 0 + self._agg_seqgap = 0 # 鏈鍐 seq 璺冲彉绱锛= 鐪熶涪鍖呮暟锛 + self._agg_rms_sum = 0.0 + self._agg_active_pkts = 0 + self._last_seq = None # 涓婁竴鍖 seq锛岀敤浜庣畻 gap + + if self.enabled: + p = Path(path or f"timing_{time.strftime('%Y%m%d_%H%M%S')}.csv") + self._f = open(p, "w", newline="", encoding="utf-8") + self._w = csv.writer(self._f) + # 缁熶竴琛ㄥご锛歬ind 鍖哄垎 grab / audio 涓ょ被琛 + self._w.writerow([ + "kind", # "grab" 鎴 "audio" + "rel_ms", # 鐩稿杩涚▼鍚姩鐨勬绉掞紙瀵归綈涓ょ被浜嬩欢鐨勬椂闂磋酱锛 + "chunk_id", # rel_ms // chunk_ms锛岀湅浜嬩欢钀藉湪绗嚑涓 chunk + "seq", + # grab 涓撶敤 + "grab_ms", # 鏈鍙栧浘/涓杞紡鏂楄楁椂 + "jpeg_bytes", + "since_last_grab_ms", + "audio_phase", # 鍙栬繖甯у浘鏃堕煶棰戞槸 active / silent锛堢浉浣嶏級 + "ms_since_voice",# 璺濇渶杩戜竴娆¢煶棰戞椿鍔ㄥ涔咃紙鎵鹃潤榛樼紳锛 + # grab 涓撶敤 鈥斺 婕忔枟瀛楁 + "reason", # 婕忔枟鍐崇瓥锛歴end/severe_shake/unstable/need_focus/orient... + "judge_ms", # run_funnel 鍒ゅ畾鑰楁椂 + "af_ms", # 瀵圭劍瑙﹀彂+settle+閲嶆姄鑰楁椂锛0=娌¤Е鍙戝鐒︼級 + "n_frames", # 鏈疆鎶撳抚鏁 + # audio 涓撶敤锛堟寜绉掕仛鍚堬級 + "audio_pps", # 杩欎竴绉掔殑鍖呮暟 + "audio_seqgap", # 杩欎竴绉 seq 璺冲彉绱 = 鐪熶涪鍖呮暟锛堜笉鏄浐浠 reserved锛 + "audio_rms", # 杩欎竴绉掑钩鍧 rms + "audio_active_pps", # 杩欎竴绉掗噷"娲诲姩"鍖呮暟 + ]) + print(f"[TimingProbe] writing {p}") + + def _rel_ms(self) -> float: + return (time.monotonic() - self._t0) * 1000.0 + + # 鈹鈹 鍙栧浘浜嬩欢 鈹鈹 + def mark_grab(self, seq: int, grab_ms: float, jpeg_bytes: int, + since_last_ms: float, reason: str = "", judge_ms=None, + af_ms=None, n_frames=None) -> None: + if not self.enabled: + return + rel = self._rel_ms() + phase = "active" if self._audio_active else "silent" + ms_since_voice = rel - self._last_audio_active_rel_ms + if ms_since_voice > 1e8: + ms_since_voice = -1 # 杩樻病鍑虹幇杩囬煶棰戞椿鍔 + self._w.writerow([ + "grab", round(rel, 1), int(rel // self.chunk_ms), seq, + round(grab_ms, 1), jpeg_bytes, round(since_last_ms, 1), + phase, round(ms_since_voice, 1), + reason, judge_ms if judge_ms is not None else "", + af_ms if af_ms is not None else "", + n_frames if n_frames is not None else "", + "", "", "", "", # audio 涓撶敤鍒楃暀绌 + ]) + self._f.flush() + + # 鈹鈹 闊抽浜嬩欢锛堟瘡鍖呰皟锛屽唴閮ㄦ寜绉掕仛鍚堝啓涓琛岋級鈹鈹 + # 娉ㄦ剰锛氬浐浠跺寘澶寸4瀛楁鏄 reserved(pad)锛屼笉鏄涪鍖呮暟锛涚湡涓㈠寘鐢 seq 璺冲彉鎺ㄦ柇銆 + def mark_audio(self, seq: int, n_samples: int, rms: float) -> None: + if not self.enabled: + return + rel = self._rel_ms() + + # 鏇存柊鐩镐綅鐘舵侊紙渚 mark_grab 璇伙級 + self._audio_active = rms > self._voice_rms_gate + if self._audio_active: + self._last_audio_active_rel_ms = rel + + # seq 璺冲彉 = 鐪熶涪鍖咃紙PC 瑙嗚锛氬簲鏀 seq 杩炵画锛岀己鍙峰嵆涓級 + gap = 0 + if self._last_seq is not None: + d = seq - self._last_seq - 1 + if 0 < d < 10000: # 鍚堢悊鑼冨洿锛屾帓闄ら噸杩/鍥炵粫 + gap = d + self._last_seq = seq + + # 鎸夌鑱氬悎 + bucket = int(rel // 1000) + if self._agg_bucket == -1: + self._agg_bucket = bucket + if bucket != self._agg_bucket: + self._flush_audio_bucket() + self._agg_bucket = bucket + self._agg_pkts += 1 + self._agg_seqgap += gap + self._agg_rms_sum += rms + if rms > self._voice_rms_gate: + self._agg_active_pkts += 1 + + def _flush_audio_bucket(self) -> None: + if self._agg_pkts == 0: + return + rel = self._agg_bucket * 1000.0 + avg_rms = self._agg_rms_sum / self._agg_pkts + self._w.writerow([ + "audio", round(rel, 1), int(rel // self.chunk_ms), "", + "", "", "", "", "", # grab 鍩虹鍒楃暀绌 + "", "", "", "", # grab 婕忔枟鍒楃暀绌 + self._agg_pkts, self._agg_seqgap, round(avg_rms, 4), self._agg_active_pkts, + ]) + self._f.flush() + self._agg_pkts = 0 + self._agg_seqgap = 0 + self._agg_rms_sum = 0.0 + self._agg_active_pkts = 0 + + def close(self) -> None: + if not self.enabled: + return + try: + self._flush_audio_bucket() + if self._f: + self._f.close() + except Exception: + pass diff --git a/extensions/assistive_harness/prompts/README.md b/extensions/assistive_harness/prompts/README.md new file mode 100644 index 0000000..38fb5ac --- /dev/null +++ b/extensions/assistive_harness/prompts/README.md @@ -0,0 +1,43 @@ +# User-editable Skill prompts + +These files are the system prompts used when the Voice Skill Harness creates a +new MiniCPM-o Duplex Session. Prompt text is read again on every activation, so +editing an existing `.txt` file takes effect on the next Skill switch without +restarting the Harness service. + +## Shipped prompts + +| Skill ID | Prompt file | Default | +|---|---|---| +| `idle_chat` | `idle_chat_zh.txt` | enabled | +| `find_object` | `find_object_zh.txt` | enabled | +| `read_text` | `read_text_zh.txt` | enabled | +| `describe_scene` | `describe_scene_zh.txt` | enabled | +| `obstacle_avoidance` | `obstacle_avoidance_zh.txt` | enabled / experimental | + +The find/read/obstacle task bodies were copied from the frozen AAAI_SI C1 +prompts. `find_object_zh.txt` additionally contains `{{target}}`, because the +new Session must receive the target extracted from the command that closed the +old Session. + +Source directory: + +`C:\Users\Lenovo\AI_Glasses_0618\AAAI_SI\submission_release\AAAI27_AISI_Anonymous_Code_Data\prompts` + +## Editing an existing prompt + +1. Back up the target `.txt` file. +2. Replace its contents with a UTF-8 system prompt. +3. Keep every declared template variable, such as `{{target}}`. +4. Switch to chat/another Skill and activate this Skill again. (The same + Skill/same slots command is intentionally deduplicated.) The new hot Session + will read the new text and record its rendered SHA-256 in telemetry. + +## Adding a Skill + +Create another `.txt` file and add a matching entry under `skills:` in +`../config/skills.example.yaml`. Configuration changes require restarting the +8021 Harness service; prompt-only changes do not. + +Obstacle avoidance has not passed a safety evaluation. Use it only for +stationary, supervised tests; never treat the Demo as a mobility safety device. diff --git a/extensions/assistive_harness/prompts/describe_scene_zh.txt b/extensions/assistive_harness/prompts/describe_scene_zh.txt new file mode 100644 index 0000000..b19376b --- /dev/null +++ b/extensions/assistive_harness/prompts/describe_scene_zh.txt @@ -0,0 +1,8 @@ +褰撳墠鎶鑳斤細鍦烘櫙鎻忚堪銆 + +瑙勫垯锛 +1. 鐢ㄤ竴鍒颁笁鍙ヨ瘽姒傛嫭褰撳墠鐢婚潰鐨勪富瑕佺墿浣撱佷綅缃叧绯诲拰鏄庢樉鐘舵併 +2. 浼樺厛鎻忚堪闂ㄥ彛銆佹闈€侀亾璺佷汉鐗┿佹樉钁楁枃瀛楀拰鍙兘褰卞搷鐢ㄦ埛琛屽姩鐨勫ぇ鐗╀綋銆 +3. 鍙弿杩板綋鍓嶅彲瑙佽瘉鎹紱鐪嬩笉娓呮椂鐩存帴璇寸敾闈笉娓呮銆 +4. 涓嶇粰鍑烘湭缁忛獙璇佺殑璺濈銆佽矾绾挎垨瀹夊叏淇濊瘉銆 +5. 閬垮厤鍙嶅鎻忚堪娌℃湁鍙樺寲鐨勫唴瀹广 diff --git a/extensions/assistive_harness/prompts/find_object_zh.txt b/extensions/assistive_harness/prompts/find_object_zh.txt new file mode 100644 index 0000000..affe3a2 --- /dev/null +++ b/extensions/assistive_harness/prompts/find_object_zh.txt @@ -0,0 +1,9 @@ +浣犳槸鏅鸿兘鐪奸暅鎵剧墿鍔╂墜銆備綘鑳界湅鍒扮敤鎴峰綋鍓嶇涓瑙嗚鐢婚潰锛屽苟鍚埌鐢ㄦ埛瑕佹壘鐨勭洰鏍囥 + +褰撳墠瀵绘壘鐩爣锛歿{target}} + +涓婅堪鐩爣宸茬粡鐢卞閮 Harness 浠庣敤鎴疯闊充腑璇嗗埆銆傛柊浼氳瘽鍚姩鍚庯紝鐩存帴瀵绘壘璇ョ洰鏍囷紝涓嶈鍐嶆璇㈤棶鐩爣鏄粈涔堛 +鐢ㄦ埛闂畬鍚庯紝蹇呴』鐩存帴鏍规嵁褰撳墠鐢婚潰鍥炵瓟锛屼笉瑕佸厛璇粹滃ソ鐨勨濃滄垜鏉ユ壘鈥濃滅◢绛夆濄 +濡傛灉鐩爣鍦ㄧ敾闈㈤噷鍙锛岀敤涓鍙ヨ瘽璇存竻瀹冪殑浣嶇疆銆佹柟鍚戝拰闄勮繎閿氱偣锛屼緥濡傗滃鍗版満鍦ㄥ乏鍓嶆柟锛岄潬杩戠獥鎴封濄 +濡傛灉鐩爣涓嶅彲瑙侊紝灏辫鈥滄病鐪嬪埌鈥濓紝骞剁粰涓涓叿浣撲笅涓姝ュ缓璁紝渚嬪杞悜銆侀潬杩戙佹姮澶存垨璋冩暣瑙掑害銆 +鍙弿杩扮敾闈腑纭疄鍙鐨勫唴瀹癸紝涓嶈鍑父璇嗙寽娴嬨傚洖绛斿敖閲忕煭銆 diff --git a/extensions/assistive_harness/prompts/idle_chat_zh.txt b/extensions/assistive_harness/prompts/idle_chat_zh.txt new file mode 100644 index 0000000..a558053 --- /dev/null +++ b/extensions/assistive_harness/prompts/idle_chat_zh.txt @@ -0,0 +1,8 @@ +浣犳槸涓鍓潰鍚戣闅滅敤鎴风殑鏈湴 AI 鐪奸暅鍔╂墜銆 + +瑙勫垯锛 +1. 鍙牴鎹綋鍓嶉煶棰戙佽棰戝拰鏄庣‘涓婁笅鏂囧洖绛旓紝涓嶈缂栭犵湅涓嶆竻鎴栨病鍚竻鐨勫唴瀹广 +2. 鍥炵瓟绠娲併佽嚜鐒讹紝浼樺厛涓鍒颁笁鍙ヨ瘽銆 +3. 涓嶈涓诲姩缁欏嚭鏈粡楠岃瘉鐨勮窛绂汇佸畨鍏ㄦ垨琛屽姩鎸囦护銆 +4. 鐢ㄦ埛鐨勨滃仠涓涓嬨侀噸鏂板紑濮嬨佹壘鐗┿佽瘑瀛椻濈瓑鎺у埗鍛戒护鐢卞閮 Harness 澶勭悊锛涗笉瑕佷簤澶烘帶鍒舵潈锛屼篃涓嶈閲嶅瑙i噴绯荤粺鏈哄埗銆 +5. 娌℃湁瓒冲璇佹嵁鏃讹紝鐩存帴璇存槑涓嶇‘瀹氾紝骞跺缓璁敤鎴峰仠绋虫垨璋冩暣瑙嗚銆 diff --git a/extensions/assistive_harness/prompts/obstacle_avoidance_zh.txt b/extensions/assistive_harness/prompts/obstacle_avoidance_zh.txt new file mode 100644 index 0000000..82cb340 --- /dev/null +++ b/extensions/assistive_harness/prompts/obstacle_avoidance_zh.txt @@ -0,0 +1,9 @@ +浣犳槸鏅鸿兘鐪奸暅閬块殰鍔╂墜銆備綘鑳界湅鍒扮敤鎴峰綋鍓嶇涓瑙嗚鐢婚潰锛屽苟鍚埌鐢ㄦ埛璇㈤棶鍓嶆柟闅滅銆 + +鐢ㄦ埛闂畬鍚庯紝蹇呴』鐩存帴鏍规嵁褰撳墠鐢婚潰鍥炵瓟锛屼笉瑕佸彧璇粹滃ソ鐨勨濃滄垜浼氱暀鎰忊濄 +濡傛灉鑳界‘璁ら殰纰嶏紝鐢ㄤ竴鍙ョ煭璇濊鏄庨殰纰嶆槸浠涔堛佷綅浜庡乏鍓嶆柟/姝e墠鏂/鍙冲墠鏂规垨鍦伴潰鍝噷锛屽苟缁欏嚭绔嬪嵆鍙墽琛岀殑瀹夊叏寤鸿锛屼緥濡傚仠涓嬨佸悜宸︾粫琛屾垨鍚戝彸缁曡銆 +鍙牴鎹敾闈腑纭疄鍙鐨勯殰纰嶅拰鍙氳绌洪棿鍒ゆ柇锛屼笉瑕佺寽娴嬶紝涓嶈鎶婃櫘閫氱墿浣撹鎶ヤ负闅滅銆 +濡傛灉鎽勫儚澶存病鏈夋湞鍚戣杩涙柟鍚戙佸湴闈笉鍙銆佺敾闈㈣閬尅鎴栬瘉鎹笉瓒筹紝鏄庣‘鍥炵瓟鈥滃綋鍓嶇敾闈笉瓒充互鍒ゆ柇鍓嶆柟鏄惁瀹夊叏鈥濓紝寤鸿鐢ㄦ埛鍏堝仠涓嬪苟鎶婇暅澶磋浆鍚戝墠鏂规垨鍦伴潰銆 +鍦ㄦ棤娉曠‘璁ら氳绌洪棿鏃讹紝涓嶅緱澹扮О鈥滃墠鏂瑰畨鍏ㄢ濇垨鐩存帴缁欏嚭閫氳鏂瑰悜銆 +濡傛灉闅滅闅忓悗杩涘叆鐢婚潰鎴栫敾闈㈣川閲忔敼鍠勶紝搴旂珛鍗虫牴鎹柊鐢婚潰鏇存柊鍒ゆ柇銆 +鍥炵瓟灏介噺绠鐭 diff --git a/extensions/assistive_harness/prompts/read_text_zh.txt b/extensions/assistive_harness/prompts/read_text_zh.txt new file mode 100644 index 0000000..5d104b6 --- /dev/null +++ b/extensions/assistive_harness/prompts/read_text_zh.txt @@ -0,0 +1,7 @@ +浣犳槸鏅鸿兘鐪奸暅璇诲瓧鍔╂墜銆備綘鑳界湅鍒扮敤鎴峰綋鍓嶇涓瑙嗚鐢婚潰锛屽苟鍚埌鐢ㄦ埛瑕佹眰璇诲瓧銆 + +鐢ㄦ埛涓寮鍙h姹傝瀛楋紝浣犲繀椤诲洖绛旓紝涓嶈兘淇濇寔娌夐粯銆 +鐢ㄦ埛闂畬鍚庯紝鐩存帴璇诲嚭鐢婚潰閲屾竻鏅板彲瑙佺殑鏂囧瓧锛屼笉瑕佸厛璇粹滃ソ鐨勨濃滄垜鏉モ濄 +鍙緭鍑轰綘鐪嬫竻鐨勬枃瀛椼佹暟瀛楁垨绗﹀彿锛涚湅涓嶆竻鐨勫瓧涓嶈鐚滐紝涓嶈鏍规嵁鐗╁搧甯歌瘑琛ュ叏銆 +濡傛灉鍙兘鐪嬫竻閮ㄥ垎鏂囧瓧锛屽彲浠ュ彧璇诲嚭鐪嬫竻鐨勯儴鍒嗐 +濡傛灉鏁翠綋鐪嬩笉娓咃紝涔熷繀椤诲洖绛斺滅湅涓嶆竻鈥濓紝骞跺缓璁潬杩戙佹瀵规枃瀛楁垨閲嶆柊瀵圭劍銆 diff --git a/extensions/assistive_harness/registry.py b/extensions/assistive_harness/registry.py new file mode 100644 index 0000000..35b2fc1 --- /dev/null +++ b/extensions/assistive_harness/registry.py @@ -0,0 +1,117 @@ +from __future__ import annotations + +import hashlib +import re +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +import yaml + + +class RegistryError(ValueError): + pass + + +@dataclass(frozen=True, slots=True) +class RenderedPrompt: + skill_id: str + text: str + path: str + sha256: str + slots: dict[str, Any] + + +class SkillRegistry: + def __init__(self, config_path: str | Path): + self.config_path = Path(config_path).resolve() + raw = yaml.safe_load(self.config_path.read_text(encoding="utf-8")) or {} + self.version = int(raw.get("version", 1)) + self.default_skill = str(raw.get("default_skill", "idle_chat")) + self.control = dict(raw.get("control") or {}) + self.skills: dict[str, dict[str, Any]] = dict(raw.get("skills") or {}) + if self.default_skill not in self.skills: + raise RegistryError(f"default skill is missing: {self.default_skill}") + self._validate_prompt_files() + + def _prompt_path(self, skill_id: str) -> Path: + spec = self.get(skill_id) + path = Path(str(spec.get("prompt_file") or "")) + if not path.is_absolute(): + path = self.config_path.parent / path + return path.resolve() + + def prompt_path(self, skill_id: str) -> Path: + """Return the user-editable prompt path exposed by the registry.""" + return self._prompt_path(skill_id) + + def _validate_prompt_files(self) -> None: + for skill_id, spec in self.skills.items(): + if not isinstance(spec, dict): + raise RegistryError(f"invalid skill spec: {skill_id}") + path = self._prompt_path(skill_id) + if not path.is_file(): + raise RegistryError(f"prompt file is missing for {skill_id}: {path}") + + def get(self, skill_id: str) -> dict[str, Any]: + spec = self.skills.get(skill_id) + if spec is None: + raise RegistryError(f"unknown skill: {skill_id}") + if not isinstance(spec, dict): + raise RegistryError(f"invalid skill spec: {skill_id}") + return spec + + def is_enabled(self, skill_id: str) -> bool: + return bool(self.get(skill_id).get("enabled", False)) + + def cooldown_ms(self, skill_id: str) -> int: + return int(self.get(skill_id).get("cooldown_ms", 0)) + + def task_trigger( + self, skill_id: str, slots: dict[str, Any] | None = None + ) -> str: + """Render the optional one-shot command sent to a fresh Skill Session.""" + text = str(self.get(skill_id).get("task_trigger") or "").strip() + slots = dict(slots or {}) + variables = set(re.findall(r"\{\{\s*([a-zA-Z_][\w]*)\s*\}\}", text)) + missing = sorted(name for name in variables if not str(slots.get(name, "")).strip()) + if missing: + raise RegistryError(f"missing task trigger variables for {skill_id}: {missing}") + for name in variables: + text = re.sub( + r"\{\{\s*" + re.escape(name) + r"\s*\}\}", + str(slots[name]), + text, + ) + return text + + def render(self, skill_id: str, slots: dict[str, Any] | None = None) -> RenderedPrompt: + if not self.is_enabled(skill_id): + raise RegistryError(f"skill is disabled: {skill_id}") + slots = dict(slots or {}) + spec = self.get(skill_id) + schema = dict(spec.get("slot_schema") or {}) + for name, slot_spec in schema.items(): + if bool((slot_spec or {}).get("required")) and not str(slots.get(name, "")).strip(): + raise RegistryError(f"missing required slot '{name}' for {skill_id}") + + path = self._prompt_path(skill_id) + text = path.read_text(encoding="utf-8") + variables = set(re.findall(r"\{\{\s*([a-zA-Z_][\w]*)\s*\}\}", text)) + missing = sorted(name for name in variables if name not in slots) + if missing: + raise RegistryError(f"missing prompt variables for {skill_id}: {missing}") + for name in variables: + text = re.sub( + r"\{\{\s*" + re.escape(name) + r"\s*\}\}", + str(slots[name]), + text, + ) + digest = hashlib.sha256(text.encode("utf-8")).hexdigest() + return RenderedPrompt( + skill_id=skill_id, + text=text, + path=str(path), + sha256=digest, + slots=slots, + ) diff --git a/extensions/assistive_harness/router.py b/extensions/assistive_harness/router.py new file mode 100644 index 0000000..43038f5 --- /dev/null +++ b/extensions/assistive_harness/router.py @@ -0,0 +1,262 @@ +from __future__ import annotations + +import re +import time +from dataclasses import dataclass + +from .registry import SkillRegistry +from .schemas import ControlEvent, ControlIntent + + +_PUNCTUATION = "锛,銆傦紒锛!?锛;锛:銆乗"'鈥溾濃樷" +_SPEECH_PARTICLES = ("鍡", "鍟", "鍛", "鍚", "鍛", "鍟", "鍝") + + +def normalize_text(text: str) -> str: + compact = re.sub(r"\s+", "", text or "").lower() + return compact.strip(_PUNCTUATION) + + +def _strip_speech_particles(text: str) -> str: + cleaned = text + while cleaned and any(cleaned.endswith(item) for item in _SPEECH_PARTICLES): + cleaned = cleaned[:-1] + return cleaned + + +def _collapse_repeated_tail(text: str, max_width: int = 4) -> str: + """Collapse a short duplicated ASR tail such as ``鎵嬫満鎵嬫満`` or ``姣嶆瘝姣峘`.""" + + cleaned = text + while cleaned: + collapsed = False + for width in range(min(max_width, len(cleaned) // 2), 0, -1): + if cleaned[-width:] == cleaned[-2 * width : -width]: + cleaned = cleaned[:-width] + collapsed = True + break + if not collapsed: + return cleaned + return cleaned + + +def _clean_slot_value(value: str) -> str: + cleaned = _strip_speech_particles(value.strip(_PUNCTUATION)) + cleaned = _collapse_repeated_tail(cleaned) + return _strip_speech_particles(cleaned) + + +@dataclass(slots=True) +class RouteResult: + intent: ControlIntent + skill_id: str | None = None + slots: dict[str, str] | None = None + reason: str = "" + confidence: float = 1.0 + + +class RuleIntentRouter: + """Small deterministic Chinese intent router with exact control commands.""" + + def __init__(self, registry: SkillRegistry): + self.registry = registry + control = registry.control + self.stop_phrases = { + normalize_text(item) + for item in (control.get("stop_speech") or {}).get("phrases", []) + } + self.stop_embedded_phrases = { + normalize_text(item) + for item in (control.get("stop_speech") or {}).get("embedded_phrases", []) + } + self.resume_embedded_phrases = { + normalize_text(item) + for item in (control.get("resume_speech") or {}).get("embedded_phrases", []) + } + self.reset_phrases = { + normalize_text(item) + for item in (control.get("reset_session") or {}).get("phrases", []) + } + self.reset_embedded_phrases = { + normalize_text(item) + for item in (control.get("reset_session") or {}).get("embedded_phrases", []) + } + self.return_phrases = { + normalize_text(item) + for item in (control.get("return_to_chat") or {}).get("phrases", []) + } + self.cancel_phrases = { + normalize_text(item) + for item in (control.get("cancel_skill") or {}).get("phrases", []) + } + + @staticmethod + def _is_explicit_command(compact: str, phrases: set[str]) -> bool: + polite_prefixes = ("楹荤儲浣犲厛", "璇", "楹荤儲", "浣犲厛", "璇蜂綘", "鍙互", "鑳戒笉鑳") + polite_suffixes = ("涓涓", "濂藉悧", "鍙互鍚", "鍟", "鍛", "鍚", "鍛", "鍟", "鍝") + + # FunASR commonly appends a sentence particle ("鍋滀竴涓嬪晩") or repeats + # the final syllable ("鍋滀竴涓嬩笅"). Build a small, bounded closure of + # command-only variants instead of doing substring matching, so a + # sentence such as "鎴戝垰鎵嶆病鏈夎鍋滀竴涓" still cannot trigger STOP. + candidates: set[str] = set() + pending = [compact] + while pending: + candidate = pending.pop() + if not candidate or candidate in candidates: + continue + candidates.add(candidate) + + for prefix in polite_prefixes: + if candidate.startswith(prefix) and len(candidate) > len(prefix): + pending.append(candidate[len(prefix) :]) + for suffix in polite_suffixes: + if candidate.endswith(suffix) and len(candidate) > len(suffix): + pending.append(candidate[: -len(suffix)]) + if len(candidate) >= 2 and candidate[-1] == candidate[-2]: + pending.append(candidate[:-1]) + + return any(candidate in phrases for candidate in candidates) + + @staticmethod + def _contains_command(compact: str, phrases: set[str]) -> bool: + """Use a literal command-anchor protocol; do not infer sentence meaning.""" + return any(phrase and phrase in compact for phrase in phrases) + + @staticmethod + def _skill_variants(compact: str) -> list[str]: + """Build bounded FunASR variants while retaining an explicit Skill verb.""" + + candidates: set[str] = set() + pending = [compact] + while pending: + candidate = pending.pop() + if not candidate or candidate in candidates: + continue + candidates.add(candidate) + + without_particles = _strip_speech_particles(candidate) + if without_particles and without_particles != candidate: + pending.append(without_particles) + + # Observed FunASR variants include ``涓涓嬩笅`` and omission of the + # pronoun in ``甯垜鎵/璇籤`. Restore only that pronoun; never invent + # a missing Skill verb such as ``鎵綻` or ``璇籤`. + collapsed_action = re.sub(r"涓涓嬩笅+", "涓涓", candidate) + if collapsed_action != candidate: + pending.append(collapsed_action) + if candidate.startswith("甯") and not candidate.startswith("甯垜"): + pending.append("甯垜" + candidate[1:]) + + for prefix in ("璇蜂綘", "楹荤儲浣", "楹荤儲", "璇", "鑳戒笉鑳"): + if candidate.startswith(prefix) and len(candidate) > len(prefix): + pending.append(candidate[len(prefix) :]) + polite_prefixes = ("璇蜂綘", "楹荤儲浣", "楹荤儲", "璇", "鑳戒笉鑳") + + def noise_rank(candidate: str) -> tuple[int, int, int, int, int, str]: + return ( + int(bool(re.search(r"涓涓嬩笅+", candidate))), + int(_strip_speech_particles(candidate) != candidate), + int(candidate.startswith("甯") and not candidate.startswith("甯垜")), + int(candidate.startswith(polite_prefixes)), + len(candidate), + candidate, + ) + + # Regexes can match both the raw and corrected ASR text. Prefer the + # bounded, lower-noise candidate so slot extraction is deterministic. + return sorted(candidates, key=noise_rank) + + def route(self, utterance: str) -> RouteResult: + compact = normalize_text(utterance) + if not compact: + return RouteResult(ControlIntent.NONE, reason="empty") + + variants = self._skill_variants(compact) + + if self._is_explicit_command(compact, self.stop_phrases): + return RouteResult(ControlIntent.STOP_SPEECH, reason="explicit stop command") + # STOP wins if a transcript contains both anchors. + if self._contains_command(compact, self.stop_embedded_phrases): + return RouteResult(ControlIntent.STOP_SPEECH, reason="stop anchor contained") + # RESET wins over RESUME and all skills. The protocol is deliberately + # literal: callers can add a product wake name to the same sentence. + if ( + self._contains_command(compact, self.reset_embedded_phrases) + or self._is_explicit_command(compact, self.reset_phrases) + ): + return RouteResult( + ControlIntent.RESET_SESSION, + skill_id=self.registry.default_skill, + slots={}, + reason=( + "reset anchor contained" + if self._contains_command(compact, self.reset_embedded_phrases) + else "explicit reset command" + ), + ) + if self._contains_command(compact, self.resume_embedded_phrases): + return RouteResult(ControlIntent.RESUME_SPEECH, reason="resume anchor contained") + if self._is_explicit_command(compact, self.cancel_phrases): + return RouteResult(ControlIntent.CANCEL_SKILL, reason="explicit cancel command") + if self._is_explicit_command(compact, self.return_phrases): + return RouteResult( + ControlIntent.RETURN_TO_CHAT, + skill_id=self.registry.default_skill, + slots={}, + reason="explicit return-to-chat command", + ) + + for skill_id, spec in self.registry.skills.items(): + if not bool(spec.get("enabled", False)) or skill_id == self.registry.default_skill: + continue + for pattern in spec.get("activation_patterns") or []: + for candidate in variants: + match = re.fullmatch(str(pattern), candidate) + if match: + slots = { + key: _clean_slot_value(value) + for key, value in match.groupdict().items() + if value + } + return RouteResult( + ControlIntent.ACTIVATE_SKILL, + skill_id=skill_id, + slots=slots, + reason=f"matched {skill_id} pattern", + ) + phrases = [normalize_text(item) for item in spec.get("activation_phrases") or []] + if any( + candidate == phrase or candidate.startswith(phrase) + for candidate in variants + for phrase in phrases + ): + return RouteResult( + ControlIntent.ACTIVATE_SKILL, + skill_id=skill_id, + slots={}, + reason=f"matched {skill_id} phrase", + ) + + return RouteResult(ControlIntent.NONE, reason="ordinary chat") + + def make_event( + self, + route: RouteResult, + *, + event_id: int, + asr_event_id: int, + utterance: str, + created_at_ms: float | None = None, + ) -> ControlEvent: + return ControlEvent( + event_id=event_id, + intent=route.intent, + skill_id=route.skill_id, + slots=dict(route.slots or {}), + utterance=utterance, + confidence=route.confidence, + asr_event_id=asr_event_id, + created_at_ms=created_at_ms if created_at_ms is not None else time.time() * 1000, + reason=route.reason, + ) diff --git a/extensions/assistive_harness/schemas.py b/extensions/assistive_harness/schemas.py new file mode 100644 index 0000000..edad067 --- /dev/null +++ b/extensions/assistive_harness/schemas.py @@ -0,0 +1,67 @@ +from __future__ import annotations + +from dataclasses import asdict, dataclass, field +from enum import Enum +from typing import Any + + +class ControlIntent(str, Enum): + STOP_SPEECH = "stop_speech" + RESUME_SPEECH = "resume_speech" + RESET_SESSION = "reset_session" + CANCEL_SKILL = "cancel_skill" + RETURN_TO_CHAT = "return_to_chat" + ACTIVATE_SKILL = "activate_skill" + NONE = "none" + + +@dataclass(slots=True) +class ASREvent: + event_id: int + utterance: str + started_at_ms: float + ended_at_ms: float + final_at_ms: float + model: str + device: str + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +@dataclass(slots=True) +class ControlEvent: + event_id: int + intent: ControlIntent + utterance: str + confidence: float + asr_event_id: int + created_at_ms: float + skill_id: str | None = None + slots: dict[str, Any] = field(default_factory=dict) + reason: str = "" + system_prompt: str | None = None + prompt_path: str | None = None + prompt_sha256: str | None = None + + def to_dict(self) -> dict[str, Any]: + payload = asdict(self) + payload["type"] = "control.intent" + payload["intent"] = self.intent.value + return payload + + +@dataclass(slots=True) +class HarnessState: + current_skill: str = "idle_chat" + current_slots: dict[str, Any] = field(default_factory=dict) + session_generation: int = 0 + control_event_id: int = 0 + asr_event_id: int = 0 + restart_in_progress: bool = False + drop_output_until_listen: bool = False + speech_hold_active: bool = False + pending_skill: dict[str, Any] | None = None + + def to_dict(self) -> dict[str, Any]: + return asdict(self) diff --git a/extensions/assistive_harness/server.py b/extensions/assistive_harness/server.py new file mode 100644 index 0000000..3d3343d --- /dev/null +++ b/extensions/assistive_harness/server.py @@ -0,0 +1,604 @@ +from __future__ import annotations + +import argparse +import asyncio +import base64 +import json +import time +import uuid +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +import numpy as np +import uvicorn +import yaml +from fastapi import FastAPI, WebSocket, WebSocketDisconnect +from fastapi.responses import JSONResponse + +from .asr.base import ASREngine +from .asr.energy_vad import EnergyVAD, UtteranceAudio +from .asr.funasr_engine import FunASREngine +from .cv.base import CVObservation, FrameEnvelope +from .cv.pipeline import CVPipeline +from .cv.registry import CVProviderRegistry, build_provider_registry +from .echo_guard import EchoGuard +from .model_log import ModelTurnAccumulator +from .registry import SkillRegistry +from .router import RuleIntentRouter +from .schemas import ASREvent, ControlEvent, ControlIntent +from .state_machine import HarnessController +from .telemetry import TelemetryWriter + + +def now_ms() -> float: + return time.time() * 1000.0 + + +def _decode_float32(value: str) -> np.ndarray: + raw = base64.b64decode(value, validate=True) + if len(raw) % 4: + raise ValueError("float32 payload length is not divisible by four") + return np.frombuffer(raw, dtype=" int: + self.event_counter += 1 + return self.event_counter + + def refresh_echo_speaking(self) -> None: + self.echo.set_ai_speaking(self.model_speaking or self.playback_active) + + def record_model_turn(self, turn: dict[str, Any]) -> None: + self.telemetry.write("model", turn) + self.telemetry.append_model_transcript(turn) + print( + "[AssistiveHarness][MODEL] " + f"turn={turn.get('turn_index')} role={turn.get('role')} " + f"generation={turn.get('generation')} skill={turn.get('skill_id')} " + f"audio_ms={turn.get('audio_ms')} text={turn.get('text')!r}", + flush=True, + ) + + async def recognize( + self, utterance: UtteranceAudio + ) -> dict[str, Any] | list[dict[str, Any]]: + async with self.asr_lock: + started = now_ms() + result = await asyncio.to_thread( + self.engine.transcribe, utterance.audio, self.vad.sample_rate + ) + return await self.route_transcript( + result.text, + confidence=result.confidence, + model=result.model, + device=result.device, + started_at_ms=utterance.started_at_ms, + ended_at_ms=utterance.ended_at_ms, + final_at_ms=now_ms(), + inference_started_at_ms=started, + ) + + def recognize_in_background(self, utterance: UtteranceAudio) -> None: + """Keep receiving microphone frames while a completed utterance is decoded.""" + + async def run() -> None: + try: + await self.outbound.put(await self.recognize(utterance)) + except asyncio.CancelledError: + raise + except Exception as exc: + await self.outbound.put( + { + "type": "error", + "code": "asr_failed", + "message": str(exc), + } + ) + + task = asyncio.create_task(run()) + self.background_tasks.add(task) + task.add_done_callback(self.background_tasks.discard) + + async def route_transcript( + self, + text: str, + *, + confidence: float, + model: str, + device: str, + started_at_ms: float | None = None, + ended_at_ms: float | None = None, + final_at_ms: float | None = None, + inference_started_at_ms: float | None = None, + ) -> dict[str, Any] | list[dict[str, Any]]: + finished = final_at_ms if final_at_ms is not None else now_ms() + asr_id = self.next_event_id() + asr_event = ASREvent( + event_id=asr_id, + utterance=text.strip(), + started_at_ms=started_at_ms if started_at_ms is not None else finished, + ended_at_ms=ended_at_ms if ended_at_ms is not None else finished, + final_at_ms=finished, + model=model, + device=device, + ) + self.telemetry.write("asr", asr_event.to_dict()) + transcript_payload: dict[str, Any] = { + "type": "asr.transcript", + "asr_event_id": asr_id, + "utterance": asr_event.utterance, + "confidence": confidence, + "final_at_ms": finished, + } + if inference_started_at_ms is not None: + self.telemetry.metric("asr_inference_ms", finished - inference_started_at_ms, asr_id) + + route = self.router.route(asr_event.utterance) + echo = self.echo.evaluate(asr_event.utterance, route.intent, at_ms=finished) + print( + "[AssistiveHarness][ASR] " + f"text={asr_event.utterance!r} intent={route.intent.value} " + f"echo={'allow' if echo.allow else 'drop'} reason={echo.reason}", + flush=True, + ) + self.telemetry.write( + "echo", + { + "asr_event_id": asr_id, + "allow": echo.allow, + "reason": echo.reason, + "similarity": echo.similarity, + }, + ) + if not echo.allow or route.intent is ControlIntent.NONE: + transcript_payload["suppressed"] = not echo.allow + transcript_payload["reason"] = echo.reason if not echo.allow else route.reason + return transcript_payload + + control_id = self.next_event_id() + event = self.router.make_event( + route, + event_id=control_id, + asr_event_id=asr_id, + utterance=asr_event.utterance, + created_at_ms=finished, + ) + decision = self.controller.process(event, now_ms=finished) + print( + "[AssistiveHarness][CONTROL] " + f"intent={route.intent.value} accepted={decision.accepted} " + f"action={decision.action} event={control_id}", + flush=True, + ) + payload = { + **decision.payload, + "accepted": decision.accepted, + "action": decision.action, + "decision_reason": decision.reason, + "client_id": self.client_id, + } + self.telemetry.write("control", payload) + if route.intent in { + ControlIntent.RESET_SESSION, + ControlIntent.CANCEL_SKILL, + ControlIntent.RETURN_TO_CHAT, + ControlIntent.ACTIVATE_SKILL, + }: + self.telemetry.write( + "skill", + { + "control_event_id": control_id, + "intent": route.intent.value, + "skill_id": event.skill_id, + "slots": event.slots, + "prompt_path": event.prompt_path, + "prompt_sha256": event.prompt_sha256, + "accepted": decision.accepted, + "action": decision.action, + }, + ) + if decision.accepted: + return [transcript_payload, payload] + transcript_payload["suppressed"] = True + transcript_payload["reason"] = decision.reason + return transcript_payload + + +class AssistiveHarnessService: + def __init__( + self, + config_path: str | Path, + *, + enabled: bool, + model_path: str | None = None, + allow_test_injection: bool = False, + engine: ASREngine | None = None, + cv_providers: CVProviderRegistry | None = None, + ): + self.config_path = Path(config_path).resolve() + self.config = yaml.safe_load(self.config_path.read_text(encoding="utf-8")) or {} + self.registry = SkillRegistry(self.config_path) + self.cv_config = dict(self.config.get("cv") or {}) + self.cv_providers = cv_providers or build_provider_registry( + self.config, config_dir=self.config_path.parent + ) + self.enabled = bool(enabled) + self.allow_test_injection = bool(allow_test_injection) + asr_cfg = dict(self.config.get("asr") or {}) + resolved_model_path = model_path or str(asr_cfg.get("model_path") or "") + if engine is not None: + self.engine = engine + elif self.enabled: + if not resolved_model_path: + raise ValueError("--model-path is required when the Harness is enabled") + self.engine = FunASREngine( + resolved_model_path, + device=str(asr_cfg.get("device") or "cpu"), + ) + else: + self.engine = None + run_id = time.strftime("%Y%m%d_%H%M%S") + "_" + uuid.uuid4().hex[:8] + telemetry_root = Path(__file__).resolve().parent / "runs" + self.telemetry = TelemetryWriter(telemetry_root, run_id, self.config) + self.clients: dict[str, ClientRuntime] = {} + self.started_at_ms = now_ms() + + def make_runtime(self, client_id: str) -> ClientRuntime: + if self.engine is None: + raise RuntimeError("Harness is disabled") + asr_cfg = dict(self.config.get("asr") or {}) + echo_cfg = dict(self.config.get("echo_guard") or {}) + return ClientRuntime( + client_id=client_id, + registry=self.registry, + router=RuleIntentRouter(self.registry), + engine=self.engine, + telemetry=self.telemetry, + vad=EnergyVAD( + rms_threshold=float(asr_cfg.get("rms_threshold", 0.012)), + min_speech_ms=int(asr_cfg.get("min_speech_ms", 180)), + end_silence_ms=int(asr_cfg.get("end_silence_ms", 450)), + max_utterance_ms=int(asr_cfg.get("max_utterance_ms", 8000)), + preroll_ms=int(asr_cfg.get("preroll_ms", 200)), + ), + echo=EchoGuard( + window_ms=int(echo_cfg.get("window_ms", 20000)), + similarity_threshold=float(echo_cfg.get("similarity_threshold", 0.86)), + ), + controller=HarnessController(self.registry), + allow_test_injection=self.allow_test_injection, + cv=CVPipeline( + self.cv_providers, + on_observation=lambda observation: self.telemetry.write( + "cv", observation.to_dict() + ), + on_metric=lambda name, value: self.telemetry.metric(name, value), + queue_size=int(self.cv_config.get("queue_size", 1)), + inference_timeout_ms=float( + self.cv_config.get("inference_timeout_ms", 2000) + ), + worker_name=f"assistive-cv-{client_id[:8]}", + ), + ) + + def create_app(self) -> FastAPI: + app = FastAPI(title="Assistive Voice Skill Harness", version="0.1.0") + + @app.get("/health") + async def health() -> JSONResponse: + return JSONResponse( + { + "ok": True, + "enabled": self.enabled, + "clients": len(self.clients), + "uptime_ms": now_ms() - self.started_at_ms, + "asr_loaded": bool(getattr(self.engine, "loaded", self.engine is not None)), + "config": self.config_path.name, + "cv_providers": sorted(self.cv_providers.providers), + } + ) + + @app.get("/skills") + async def skills() -> JSONResponse: + return JSONResponse( + { + "ok": True, + "default_skill": self.registry.default_skill, + "config_path": str(self.config_path), + "skills": { + skill_id: { + "enabled": self.registry.is_enabled(skill_id), + "description": str(spec.get("description") or ""), + "prompt_path": str(self.registry.prompt_path(skill_id)), + "requires_session_restart": bool( + spec.get("requires_session_restart", True) + ), + "cv_mode": str(spec.get("cv_mode") or "disabled"), + "cv_provider": str(spec.get("cv_provider") or "noop"), + } + for skill_id, spec in self.registry.skills.items() + }, + } + ) + + @app.on_event("shutdown") + async def write_shutdown_summary() -> None: + self.telemetry.write_summary( + { + "enabled": self.enabled, + "uptime_ms": now_ms() - self.started_at_ms, + "connected_clients_at_shutdown": len(self.clients), + "asr_loaded": bool(getattr(self.engine, "loaded", self.engine is not None)), + } + ) + + @app.websocket("/ws/control") + async def control_socket(websocket: WebSocket) -> None: + if not self.enabled: + await websocket.close(code=1013, reason="Harness disabled") + return + await websocket.accept() + client_id = websocket.query_params.get("client_id") or uuid.uuid4().hex + runtime = self.make_runtime(client_id) + self.clients[client_id] = runtime + + async def send_outbound() -> None: + while True: + response = await runtime.outbound.get() + if isinstance(response, list): + for item in response: + await websocket.send_json(item) + elif response is not None: + await websocket.send_json(response) + + sender = asyncio.create_task(send_outbound()) + await runtime.outbound.put( + { + "type": "harness.ready", + "client_id": client_id, + "state": runtime.controller.state.to_dict(), + } + ) + try: + while True: + message = await websocket.receive_json() + response = await self._handle_message(runtime, message) + if response is not None: + await runtime.outbound.put(response) + except WebSocketDisconnect: + runtime.controller.mark_disconnected() + finally: + pending_turn = runtime.model_turns.flush() + if pending_turn is not None: + runtime.record_model_turn(pending_turn) + sender.cancel() + for task in tuple(runtime.background_tasks): + task.cancel() + await runtime.cv.close() + await asyncio.gather(sender, *runtime.background_tasks, return_exceptions=True) + self.clients.pop(client_id, None) + + return app + + async def _handle_message( + self, runtime: ClientRuntime, message: dict[str, Any] + ) -> dict[str, Any] | list[dict[str, Any]] | None: + message_type = str(message.get("type") or "") + if message_type == "ping": + return {"type": "pong", "at_ms": now_ms()} + if message_type == "audio.mirror": + audio = _decode_float32(str(message.get("audio_b64") or "")) + utterance = runtime.vad.feed(audio, float(message.get("started_at_ms") or now_ms())) + if utterance is None: + return None + runtime.recognize_in_background(utterance) + return None + if message_type == "asr.inject": + if not runtime.allow_test_injection: + return {"type": "error", "code": "test_injection_disabled"} + return await runtime.route_transcript( + str(message.get("text") or ""), + confidence=1.0, + model="injected-test-only", + device="none", + ) + if message_type in ("funnel.stop", "funnel.resume"): + # 婕忔枟 reject/鎭㈠:绋嬪簭鐩存帴瑙﹀彂,涓嶇粡 ASR/router 鏂囧瓧鍖归厤銆 + # 澶嶇敤 controller.process + STOP_SPEECH/RESUME_SPEECH,鍜岀湡浜恒屽仠涓涓/鎭㈠瀵硅瘽銆 + # 璧板畬鍏ㄧ浉鍚岀殑涓嬫父(璁惧绔敹鍒 control.intent 鎵ц stop_speech/resume_speech)銆 + # 褰掑彛 8021:telemetry 缁熶竴璁板綍,鏍 source=funnel 浠ュ尯鍒嗙湡浜鸿繕鏄紡鏂楄Е鍙戙 + is_stop = message_type == "funnel.stop" + intent = ControlIntent.STOP_SPEECH if is_stop else ControlIntent.RESUME_SPEECH + event = ControlEvent( + event_id=runtime.next_event_id(), + intent=intent, + utterance="[%s]" % message_type, + confidence=1.0, + asr_event_id=runtime.next_event_id(), + created_at_ms=now_ms(), + reason=str(message.get("reason") or "funnel"), + ) + decision = runtime.controller.process(event, now_ms=event.created_at_ms) + payload = { + **decision.payload, + "accepted": decision.accepted, + "action": decision.action, + "decision_reason": decision.reason, + "source": "funnel", + "client_id": runtime.client_id, + } + runtime.telemetry.write("control", payload) + print( + "[AssistiveHarness][FUNNEL] " + "type=%s accepted=%s action=%s reason=%s" + % (message_type, decision.accepted, decision.action, event.reason), + flush=True, + ) + return payload if decision.accepted else None + if message_type == "model.state": + event = dict(message) + runtime.telemetry.write("model", event) + speaking = str(message.get("state") or "") == "speak" + runtime.model_speaking = speaking + runtime.refresh_echo_speaking() + text = str(message.get("text") or "") + if speaking and text: + runtime.echo.note_model_text(text) + for turn in runtime.model_turns.feed(event): + runtime.record_model_turn(turn) + return None + if message_type == "playback.state": + event = dict(message) + runtime.telemetry.write("model", event) + runtime.playback_active = bool(message.get("active")) + runtime.refresh_echo_speaking() + return None + if message_type == "session.state": + event = dict(message) + runtime.telemetry.write("session", event) + phase = str(message.get("phase") or "") + if phase == "listen": + runtime.controller.mark_listen_fence() + elif phase == "restart_complete": + runtime.controller.mark_restart_complete( + str(message.get("skill_id") or runtime.registry.default_skill), + dict(message.get("slots") or {}), + int(message.get("generation") or 0), + ) + return None + if message_type == "control.ack": + if bool(message.get("ok")) and str(message.get("intent") or "") == "stop_speech": + # Browser STOP has already flushed/blocked playback. Do not + # leave EchoGuard in a stale speaking state while the user + # immediately issues the next explicit command. + runtime.model_speaking = False + runtime.playback_active = False + runtime.refresh_echo_speaking() + runtime.telemetry.write("session", dict(message)) + print( + "[AssistiveHarness][ACK] " + f"intent={message.get('intent')} ok={message.get('ok')} " + f"generation={message.get('generation')} " + f"session={message.get('old_session_id') or '-'}" + f"->{message.get('new_session_id') or '-'} " + f"cleanup={message.get('cleanup_mode') or '-'} " + f"restart_ms={message.get('restart_latency_ms') or '-'} " + f"stale_text={message.get('dropped_old_text') or 0} " + f"stale_audio={message.get('dropped_old_audio') or 0}", + flush=True, + ) + metric = message.get("restart_latency_ms") + if isinstance(metric, (int, float)): + runtime.telemetry.metric( + "restart_latency_ms", float(metric), int(message.get("event_id") or 0) + ) + return None + if message_type == "frame.shadow": + at_ms = float(message.get("timestamp_ms") or now_ms()) + max_fps = max(0.0, float(self.cv_config.get("max_fps", 1))) + if max_fps <= 0: + return None + if at_ms - runtime.last_frame_ms < 1000.0 / max_fps: + return None + skill_id = runtime.controller.state.current_skill + skill_spec = runtime.registry.get(skill_id) + mode = str(skill_spec.get("cv_mode") or self.cv_config.get("mode") or "disabled") + if mode == "disabled": + return None + provider_id = str(skill_spec.get("cv_provider") or "noop") + frame_b64 = str(message.get("jpeg_b64") or "") + frame_id = str(message.get("frame_id") or uuid.uuid4().hex) + try: + frame = base64.b64decode(frame_b64, validate=True) if frame_b64 else None + except Exception as exc: + runtime.telemetry.write( + "cv", + CVObservation( + frame_id=frame_id, + timestamp_ms=at_ms, + skill_id=skill_id, + provider=f"{mode}:{provider_id}", + values={"status": "error"}, + error=f"InvalidFramePayload: {exc}", + ).to_dict(), + ) + return None + runtime.last_frame_ms = at_ms + runtime.cv.submit( + FrameEnvelope( + frame=frame, + frame_id=frame_id, + timestamp_ms=at_ms, + skill_id=skill_id, + slots=dict(runtime.controller.state.current_slots), + mode=mode, + provider_id=provider_id, + ) + ) + return None + return {"type": "error", "code": "unknown_message", "message_type": message_type} + + +def build_parser() -> argparse.ArgumentParser: + default_config = Path(__file__).resolve().parent / "config" / "skills.example.yaml" + parser = argparse.ArgumentParser(description="Optional local assistive voice-skill Harness") + parser.add_argument("--config", default=str(default_config)) + parser.add_argument("--enabled", action="store_true", help="explicit opt-in; default is off") + parser.add_argument("--host", default="127.0.0.1") + parser.add_argument("--port", type=int, default=8021) + parser.add_argument("--model-path", default=None) + parser.add_argument("--allow-test-injection", action="store_true") + parser.add_argument("--certfile", default=None) + parser.add_argument("--keyfile", default=None) + return parser + + +def main() -> None: + args = build_parser().parse_args() + service = AssistiveHarnessService( + args.config, + enabled=args.enabled, + model_path=args.model_path, + allow_test_injection=args.allow_test_injection, + ) + if isinstance(service.engine, FunASREngine): + warmup_started = time.perf_counter() + print("[AssistiveHarness] loading and warming FunASR before accepting clients...") + service.engine.warm_up() + print( + "[AssistiveHarness] FunASR ready " + f"({time.perf_counter() - warmup_started:.2f}s)" + ) + uvicorn.run( + service.create_app(), + host=args.host, + port=args.port, + ssl_certfile=args.certfile, + ssl_keyfile=args.keyfile, + ) + + +if __name__ == "__main__": + main() diff --git a/extensions/assistive_harness/state_machine.py b/extensions/assistive_harness/state_machine.py new file mode 100644 index 0000000..e0cd9ce --- /dev/null +++ b/extensions/assistive_harness/state_machine.py @@ -0,0 +1,115 @@ +from __future__ import annotations + +import time +from dataclasses import dataclass +from typing import Any + +from .registry import RegistryError, SkillRegistry +from .schemas import ControlEvent, ControlIntent, HarnessState + + +@dataclass(slots=True) +class ControlDecision: + accepted: bool + action: str + reason: str + payload: dict[str, Any] + + +class HarnessController: + def __init__(self, registry: SkillRegistry): + self.registry = registry + self.state = HarnessState(current_skill=registry.default_skill) + self._seen_asr_ids: set[int] = set() + self._last_skill_activation_ms: dict[tuple[str, tuple[tuple[str, str], ...]], float] = {} + + @staticmethod + def _skill_key(skill_id: str, slots: dict[str, Any]) -> tuple[str, tuple[tuple[str, str], ...]]: + return skill_id, tuple(sorted((str(key), str(value)) for key, value in slots.items())) + + def process(self, event: ControlEvent, now_ms: float | None = None) -> ControlDecision: + now = now_ms if now_ms is not None else time.time() * 1000 + if event.asr_event_id in self._seen_asr_ids: + return ControlDecision(False, "ignore", "duplicate_asr_event", {}) + self._seen_asr_ids.add(event.asr_event_id) + self.state.asr_event_id = max(self.state.asr_event_id, event.asr_event_id) + self.state.control_event_id = max(self.state.control_event_id, event.event_id) + + if event.intent is ControlIntent.STOP_SPEECH: + self.state.speech_hold_active = True + self.state.drop_output_until_listen = True + return ControlDecision(True, "stop_speech", "highest_priority", event.to_dict()) + + if event.intent is ControlIntent.RESUME_SPEECH: + self.state.speech_hold_active = False + self.state.drop_output_until_listen = False + return ControlDecision(True, "resume_speech", "explicit_resume", event.to_dict()) + + if event.intent is ControlIntent.RESET_SESSION: + rendered = self.registry.render(self.registry.default_skill, {}) + event.skill_id = self.registry.default_skill + event.slots = {} + event.system_prompt = rendered.text + event.prompt_path = rendered.path + event.prompt_sha256 = rendered.sha256 + self.state.restart_in_progress = True + self.state.pending_skill = None + return ControlDecision(True, "restart_session", "explicit_reset", event.to_dict()) + + if event.intent in {ControlIntent.CANCEL_SKILL, ControlIntent.RETURN_TO_CHAT}: + event.skill_id = self.registry.default_skill + event.slots = {} + + if event.intent in { + ControlIntent.CANCEL_SKILL, + ControlIntent.RETURN_TO_CHAT, + ControlIntent.ACTIVATE_SKILL, + }: + skill_id = event.skill_id or self.registry.default_skill + slots = dict(event.slots or {}) + try: + rendered = self.registry.render(skill_id, slots) + except RegistryError as exc: + return ControlDecision(False, "clarify", str(exc), event.to_dict()) + + if self.state.restart_in_progress: + self.state.pending_skill = {"skill_id": skill_id, "slots": slots} + return ControlDecision(True, "queue_skill", "restart_in_progress_last_write_wins", event.to_dict()) + + key = self._skill_key(skill_id, slots) + last = self._last_skill_activation_ms.get(key) + cooldown = self.registry.cooldown_ms(skill_id) + if skill_id == self.state.current_skill and slots == self.state.current_slots: + return ControlDecision(False, "ignore", "same_skill_same_slots", event.to_dict()) + if last is not None and now - last < cooldown: + return ControlDecision(False, "ignore", "skill_cooldown", event.to_dict()) + + self._last_skill_activation_ms[key] = now + event.system_prompt = rendered.text + event.prompt_path = rendered.path + event.prompt_sha256 = rendered.sha256 + self.state.restart_in_progress = True + return ControlDecision(True, "activate_skill", "skill_switch", event.to_dict()) + + return ControlDecision(False, "pass_chat", "ordinary_chat", event.to_dict()) + + def mark_restart_complete(self, skill_id: str, slots: dict[str, Any], generation: int) -> dict[str, Any] | None: + self.state.current_skill = skill_id + self.state.current_slots = dict(slots) + self.state.session_generation = int(generation) + self.state.restart_in_progress = False + self.state.speech_hold_active = False + self.state.drop_output_until_listen = False + pending = self.state.pending_skill + self.state.pending_skill = None + return pending + + def mark_listen_fence(self) -> None: + if not self.state.speech_hold_active: + self.state.drop_output_until_listen = False + + def mark_disconnected(self) -> None: + self.state.restart_in_progress = False + self.state.pending_skill = None + self.state.speech_hold_active = False + self.state.drop_output_until_listen = False diff --git a/extensions/assistive_harness/summarize_reset_run.py b/extensions/assistive_harness/summarize_reset_run.py new file mode 100644 index 0000000..1e8ce19 --- /dev/null +++ b/extensions/assistive_harness/summarize_reset_run.py @@ -0,0 +1,97 @@ +from __future__ import annotations + +import argparse +import json +import math +from pathlib import Path +from typing import Any + + +def nearest_rank(values: list[float], quantile: float) -> float | None: + if not values: + return None + ordered = sorted(values) + index = max(0, math.ceil(len(ordered) * quantile) - 1) + return ordered[index] + + +def load_events(path: Path) -> list[dict[str, Any]]: + if not path.exists(): + return [] + events: list[dict[str, Any]] = [] + for line in path.read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + try: + events.append(json.loads(line)) + except json.JSONDecodeError: + continue + return events + + +def resolve_run(value: str | None) -> Path: + runs_root = Path(__file__).resolve().parent / "runs" + if value: + candidate = Path(value).expanduser().resolve() + if not candidate.is_dir(): + raise SystemExit(f"Run directory does not exist: {candidate}") + return candidate + candidates = [path for path in runs_root.iterdir() if path.is_dir()] + if not candidates: + raise SystemExit(f"No run directories found under {runs_root}") + return max(candidates, key=lambda path: path.stat().st_mtime) + + +def main() -> None: + parser = argparse.ArgumentParser(description="Summarize real browser RESET telemetry") + parser.add_argument("--run", help="specific run directory; defaults to the newest run") + args = parser.parse_args() + + run_dir = resolve_run(args.run) + events = load_events(run_dir / "session_events.jsonl") + resets = [ + event + for event in events + if event.get("type") == "control.ack" + and event.get("intent") == "reset_session" + ] + successes = [event for event in resets if event.get("ok") is True] + latencies = [ + float(event["restart_latency_ms"]) + for event in successes + if isinstance(event.get("restart_latency_ms"), (int, float)) + ] + changed_ids = [ + event + for event in successes + if event.get("old_session_id") + and event.get("new_session_id") + and event.get("old_session_id") != event.get("new_session_id") + ] + + print(f"Run: {run_dir}") + print(f"RESET success: {len(successes)}/{len(resets)}") + print(f"Session ID changed: {len(changed_ids)}/{len(successes)}") + if latencies: + print(f"Restart P50: {nearest_rank(latencies, 0.50):.1f} ms") + print(f"Restart P90: {nearest_rank(latencies, 0.90):.1f} ms") + print(f"Restart max: {max(latencies):.1f} ms") + else: + print("Restart P50/P90: no RESET latency samples") + print( + "Dropped stale callbacks (cumulative): " + f"text={max((int(event.get('dropped_old_text') or 0) for event in successes), default=0)}, " + f"audio={max((int(event.get('dropped_old_audio') or 0) for event in successes), default=0)}" + ) + print("Audible old-output pollution: manual observation required") + for index, event in enumerate(successes, start=1): + print( + f"{index:02d}. generation={event.get('generation')} " + f"session={event.get('old_session_id') or '-'}->{event.get('new_session_id') or '-'} " + f"cleanup={event.get('cleanup_mode') or '-'} " + f"latency={float(event.get('restart_latency_ms') or 0):.1f} ms" + ) + + +if __name__ == "__main__": + main() diff --git a/extensions/assistive_harness/telemetry.py b/extensions/assistive_harness/telemetry.py new file mode 100644 index 0000000..2e3aa18 --- /dev/null +++ b/extensions/assistive_harness/telemetry.py @@ -0,0 +1,67 @@ +from __future__ import annotations + +import csv +import json +import threading +import time +from pathlib import Path +from typing import Any + +import yaml + + +class TelemetryWriter: + FILES = { + "asr": "asr_events.jsonl", + "control": "control_events.jsonl", + "session": "session_events.jsonl", + "skill": "skill_events.jsonl", + "cv": "cv_events.jsonl", + "echo": "echo_events.jsonl", + "model": "model_events.jsonl", + } + + def __init__(self, root: str | Path, run_id: str, config: dict[str, Any]): + self.run_id = run_id + self.run_dir = Path(root) / run_id + self.run_dir.mkdir(parents=True, exist_ok=True) + self._lock = threading.Lock() + (self.run_dir / "config_snapshot.yaml").write_text( + yaml.safe_dump(config, allow_unicode=True, sort_keys=True), + encoding="utf-8", + ) + self._metrics_path = self.run_dir / "metrics.csv" + with self._metrics_path.open("w", newline="", encoding="utf-8") as handle: + csv.writer(handle).writerow(["timestamp_ms", "metric", "value", "event_id"]) + + def write(self, stream: str, payload: dict[str, Any]) -> None: + filename = self.FILES[stream] + record = {"logged_at_ms": time.time() * 1000, **payload} + line = json.dumps(record, ensure_ascii=False, sort_keys=True) + with self._lock: + with (self.run_dir / filename).open("a", encoding="utf-8") as handle: + handle.write(line + "\n") + + def metric(self, name: str, value: float, event_id: int | None = None) -> None: + with self._lock: + with self._metrics_path.open("a", newline="", encoding="utf-8") as handle: + csv.writer(handle).writerow([time.time() * 1000, name, value, event_id or ""]) + + def write_summary(self, summary: dict[str, Any]) -> None: + (self.run_dir / "summary.json").write_text( + json.dumps(summary, ensure_ascii=False, indent=2, sort_keys=True), + encoding="utf-8", + ) + + def append_model_transcript(self, turn: dict[str, Any]) -> None: + text = str(turn.get("text") or "").replace("\r", " ").replace("\n", " ") + line = ( + f"turn={turn.get('turn_index')} role={turn.get('role')} " + f"generation={turn.get('generation')} skill={turn.get('skill_id')} " + f"audio_ms={turn.get('audio_ms')} text={text}\n" + ) + with self._lock: + with (self.run_dir / "model_transcript.txt").open( + "a", encoding="utf-8" + ) as handle: + handle.write(line) diff --git a/extensions/assistive_harness/tests/__init__.py b/extensions/assistive_harness/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/extensions/assistive_harness/tests/test_core.py b/extensions/assistive_harness/tests/test_core.py new file mode 100644 index 0000000..83066c9 --- /dev/null +++ b/extensions/assistive_harness/tests/test_core.py @@ -0,0 +1,402 @@ +from __future__ import annotations + +import asyncio +import tempfile +import time +import unittest +from pathlib import Path + +import numpy as np + +from extensions.assistive_harness.asr.energy_vad import EnergyVAD +from extensions.assistive_harness.cv.noop import ShadowCVProvider +from extensions.assistive_harness.cv.base import CVObservation, FrameEnvelope +from extensions.assistive_harness.cv.pipeline import CVPipeline +from extensions.assistive_harness.cv.registry import CVProviderRegistry +from extensions.assistive_harness.echo_guard import EchoGuard +from extensions.assistive_harness.model_log import ModelTurnAccumulator +from extensions.assistive_harness.registry import RegistryError, SkillRegistry +from extensions.assistive_harness.router import RuleIntentRouter +from extensions.assistive_harness.schemas import ControlIntent +from extensions.assistive_harness.state_machine import HarnessController + + +CONFIG = Path(__file__).resolve().parents[1] / "config" / "skills.example.yaml" + + +class RouterCorpusTests(unittest.TestCase): + @classmethod + def setUpClass(cls) -> None: + cls.registry = SkillRegistry(CONFIG) + cls.router = RuleIntentRouter(cls.registry) + + def test_control_and_skill_corpus_over_fifty_utterances(self) -> None: + corpus = [ + ("鍋滀竴涓", ControlIntent.STOP_SPEECH, None), + ("鍋滀竴涓嬪晩", ControlIntent.STOP_SPEECH, None), + ("鍋滀竴涓嬩笅", ControlIntent.STOP_SPEECH, None), + ("鍚屼竴涓", ControlIntent.STOP_SPEECH, None), + ("鍚屼竴涓嬩笅", ControlIntent.STOP_SPEECH, None), + ("绛変竴涓", ControlIntent.STOP_SPEECH, None), + ("绛変竴涓涓", ControlIntent.STOP_SPEECH, None), + ("璇峰仠涓涓嬪晩", ControlIntent.STOP_SPEECH, None), + ("璇峰仠涓涓", ControlIntent.STOP_SPEECH, None), + ("楹荤儲浣犲厛鍋滀竴涓", ControlIntent.STOP_SPEECH, None), + ("鍒浜", ControlIntent.STOP_SPEECH, None), + ("璇蜂綘鍒浜嗗ソ鍚", ControlIntent.STOP_SPEECH, None), + ("闂槾", ControlIntent.STOP_SPEECH, None), + ("鍋滄鎾姤", ControlIntent.STOP_SPEECH, None), + ("瀹夐潤", ControlIntent.STOP_SPEECH, None), + ("浣犲厛鍋滀竴涓嬶紝鎴戞湁涓棶棰", ControlIntent.STOP_SPEECH, None), + ("璇村緱鏈夌偣闀夸簡楹荤儲鍋滀竴涓嬪惂", ControlIntent.STOP_SPEECH, None), + ("鐜板湪鍙互鍏堝仠涓涓嬬劧鍚庡惉鎴戣鍚", ControlIntent.STOP_SPEECH, None), + ("涓嶅仠鍋滀竴涓嬬户缁粙缁", ControlIntent.STOP_SPEECH, None), + ("鎴戝垰鎵嶆病鏈夎鍋滀竴涓", ControlIntent.STOP_SPEECH, None), + ("璇蜂笉瑕佸仠涓涓嬶紝缁х画浠嬬粛", ControlIntent.STOP_SPEECH, None), + ("鍋滀竴涓嬫槸浠涔堟剰鎬", ControlIntent.STOP_SPEECH, None), + ("鎭㈠瀵硅瘽", ControlIntent.RESUME_SPEECH, None), + ("濂戒簡鐜板湪鎭㈠瀵硅瘽鍚", ControlIntent.RESUME_SPEECH, None), + ("涔愬鎭㈠瀵硅瘽", ControlIntent.RESUME_SPEECH, None), + ("鍋滀竴涓嬬劧鍚庢仮澶嶅璇", ControlIntent.STOP_SPEECH, None), + ("閲嶆柊寮濮", ControlIntent.RESET_SESSION, "idle_chat"), + ("璇烽噸鏂板紑濮", ControlIntent.RESET_SESSION, "idle_chat"), + ("閲嶇疆浼氳瘽", ControlIntent.NONE, None), + ("涔愬璇烽噸鏂板紑濮", ControlIntent.RESET_SESSION, "idle_chat"), + ("浣犳湁鐐瑰崱浜嗚閲嶆柊寮濮嬩細璇", ControlIntent.RESET_SESSION, "idle_chat"), + ("璇风幇鍦ㄩ噸缃細璇濈劧鍚庡惉鎴戣", ControlIntent.NONE, None), + ("鍋滀竴涓嬬劧鍚庨噸鏂板紑濮", ControlIntent.STOP_SPEECH, None), + ("鎭㈠瀵硅瘽鐒跺悗閲嶆柊寮濮", ControlIntent.RESET_SESSION, "idle_chat"), + ("鏂板缓浼氳瘽", ControlIntent.NONE, None), + ("娓呯┖涓婁笅鏂", ControlIntent.NONE, None), + ("鍥炲埌鑱婂ぉ", ControlIntent.RETURN_TO_CHAT, "idle_chat"), + ("璇峰洖鍒拌亰澶", ControlIntent.RETURN_TO_CHAT, "idle_chat"), + ("鍥炲埌鏅氳亰澶", ControlIntent.RETURN_TO_CHAT, "idle_chat"), + ("鍥炲埌鏅氳亰澶╁ぉ", ControlIntent.RETURN_TO_CHAT, "idle_chat"), + ("閫鍑烘妧鑳", ControlIntent.RETURN_TO_CHAT, "idle_chat"), + ("鏅氳亰澶", ControlIntent.RETURN_TO_CHAT, "idle_chat"), + ("鍙栨秷浠诲姟", ControlIntent.CANCEL_SKILL, None), + ("涓嶆壘浜", ControlIntent.CANCEL_SKILL, None), + ("涓嶈浜", ControlIntent.CANCEL_SKILL, None), + ("甯垜鎵炬墜鏈", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("甯垜鎵句竴涓嬫垜鐨勬墜鏈", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("甯垜鎵句竴涓嬫墜鏈", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("甯垜鎵句竴涓嬩笅鎴戠殑鎵嬫満鍡", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("甯壘涓涓嬫墜鏈烘墜鏈", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("璇峰府鎴戞壘閽ュ寵", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("姘存澂鍦ㄥ摢", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("涔﹀湪鍝噷", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("鐪嬪埌闂ㄥ崱浜嗗悧", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("鏈夋病鏈夐洦浼", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("楹荤儲甯垜鎵剧溂闀", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("璇烽棶閽卞寘鍦ㄥ摢", ControlIntent.ACTIVATE_SKILL, "find_object"), + ("璇讳竴涓", ControlIntent.ACTIVATE_SKILL, "read_text"), + ("璇讳竴涓嬭繖琛屽瓧", ControlIntent.ACTIVATE_SKILL, "read_text"), + ("璇疯涓涓", ControlIntent.ACTIVATE_SKILL, "read_text"), + ("甯垜璇讳竴涓嬭繖涓澂瀛愪笂鐨勫瓧姣嶆瘝姣", ControlIntent.ACTIVATE_SKILL, "read_text"), + ("甯涓涓嬭繖涓按鏉笂闈㈢殑瀛楀瓧涓", ControlIntent.ACTIVATE_SKILL, "read_text"), + ("甯垜璇嗗瓧", ControlIntent.ACTIVATE_SKILL, "read_text"), + ("涓婇潰鍐欎簡浠涔", ControlIntent.ACTIVATE_SKILL, "read_text"), + ("杩欐槸浠涔堝瓧", ControlIntent.ACTIVATE_SKILL, "read_text"), + ("璇绘枃瀛", ControlIntent.ACTIVATE_SKILL, "read_text"), + ("璇蜂綘璇绘枃瀛", ControlIntent.ACTIVATE_SKILL, "read_text"), + ("鎻忚堪涓涓", ControlIntent.ACTIVATE_SKILL, "describe_scene"), + ("甯垜鎻忚堪涓涓", ControlIntent.ACTIVATE_SKILL, "describe_scene"), + ("鐪嬬湅鍛ㄥ洿", ControlIntent.ACTIVATE_SKILL, "describe_scene"), + ("甯垜閬块殰", ControlIntent.ACTIVATE_SKILL, "obstacle_avoidance"), + ("鍓嶉潰鏈夐殰纰嶅悧", ControlIntent.ACTIVATE_SKILL, "obstacle_avoidance"), + ("浠婂ぉ澶╂皵鎬庝箞鏍", ControlIntent.NONE, None), + ("鍋滄鏄竴涓姩璇", ControlIntent.NONE, None), + ("浠栬璁╂垜闂槾浣嗘垜娌″悓鎰", ControlIntent.NONE, None), + ("閲嶆柊寮濮嬭繖涓瘝鎬庝箞缈昏瘧", ControlIntent.RESET_SESSION, "idle_chat"), + ("鎴戝湪璇讳竴鏈功", ControlIntent.NONE, None), + ("浣犺寰楄繖鏉按鎬庝箞鏍", ControlIntent.NONE, None), + ("鏅氳亰澶╂満鍣ㄤ汉鏄粈涔", ControlIntent.NONE, None), + ("璇蜂粙缁嶄竴涓嬩笂娴", ControlIntent.NONE, None), + ("鎴戜笉鎯虫竻绌轰笂涓嬫枃鍥犱负杩樻湁鐢", ControlIntent.NONE, None), + ("甯垜鍒嗘瀽杩欐璇", ControlIntent.NONE, None), + ("甯垜涓涓嬫墜鏈烘墜鏈", ControlIntent.NONE, None), + ("鑳藉惉瑙佹垜鍚", ControlIntent.NONE, None), + ("浠婂ぉ鍛ㄥ嚑", ControlIntent.NONE, None), + ("涓轰粈涔堜細杩欐牱", ControlIntent.NONE, None), + ("缁х画璇", ControlIntent.NONE, None), + ("鎺ョ潃璇", ControlIntent.NONE, None), + ("浣犵户缁璇濆惂", ControlIntent.NONE, None), + ("璋㈣阿", ControlIntent.NONE, None), + ("浣犲ソ", ControlIntent.NONE, None), + ("宸﹁竟鏈変粈涔", ControlIntent.NONE, None), + ("缁欐垜璁蹭釜绗戣瘽", ControlIntent.NONE, None), + ("", ControlIntent.NONE, None), + ("銆", ControlIntent.NONE, None), + ("瀹夐潤鏄竴绉嶇姸鎬", ControlIntent.NONE, None), + ("鍙栨秷浠诲姟鏄笉鏄竴涓寜閽", ControlIntent.NONE, None), + ("鏈変汉璇翠笉璇讳簡鐒跺悗绂诲紑", ControlIntent.NONE, None), + ] + self.assertGreaterEqual(len(corpus), 50) + for utterance, expected_intent, expected_skill in corpus: + with self.subTest(utterance=utterance): + result = self.router.route(utterance) + self.assertEqual(result.intent, expected_intent) + self.assertEqual(result.skill_id, expected_skill) + + def test_find_target_slot(self) -> None: + result = self.router.route("璇峰府鎴戞壘娣辩豢鑹茬殑涔") + self.assertEqual(result.slots, {"target": "娣辩豢鑹茬殑涔"}) + + natural = self.router.route("甯垜鎵句竴涓嬫垜鐨勬墜鏈") + self.assertEqual(natural.slots, {"target": "鎵嬫満"}) + + repeated = self.router.route("甯垜鎵句竴涓嬩笅鎴戠殑鎵嬫満鍡") + self.assertEqual(repeated.slots, {"target": "鎵嬫満"}) + + missing_pronoun = self.router.route("甯壘涓涓嬫墜鏈烘墜鏈") + self.assertEqual(missing_pronoun.slots, {"target": "鎵嬫満"}) + + +class RegistryAndStateTests(unittest.TestCase): + def setUp(self) -> None: + self.registry = SkillRegistry(CONFIG) + self.router = RuleIntentRouter(self.registry) + self.controller = HarnessController(self.registry) + + def event(self, text: str, event_id: int, asr_id: int): + return self.router.make_event( + self.router.route(text), event_id=event_id, asr_event_id=asr_id, + utterance=text, created_at_ms=float(event_id * 1000), + ) + + def test_prompt_render_and_hash(self) -> None: + prompt = self.registry.render("find_object", {"target": "鎵嬫満"}) + self.assertIn("褰撳墠瀵绘壘鐩爣锛氭墜鏈", prompt.text) + self.assertIn("鏅鸿兘鐪奸暅鎵剧墿鍔╂墜", prompt.text) + self.assertEqual(len(prompt.sha256), 64) + self.assertIn("鎵嬫満", self.registry.task_trigger("find_object", {"target": "鎵嬫満"})) + + def test_aaai_skill_prompt_files_are_user_editable_and_registered(self) -> None: + find_prompt = self.registry.render("find_object", {"target": "姘存澂"}) + read_prompt = self.registry.render("read_text", {}) + self.assertTrue(Path(find_prompt.path).is_file()) + self.assertTrue(Path(read_prompt.path).is_file()) + self.assertIn("褰撳墠瀵绘壘鐩爣锛氭按鏉", find_prompt.text) + self.assertIn("鏅鸿兘鐪奸暅璇诲瓧鍔╂墜", read_prompt.text) + self.assertTrue(self.registry.is_enabled("describe_scene")) + self.assertTrue(self.registry.is_enabled("obstacle_avoidance")) + self.assertTrue( + Path(self.registry.get("obstacle_avoidance")["prompt_file"]).name + == "obstacle_avoidance_zh.txt" + ) + + def test_required_slot_validation(self) -> None: + with self.assertRaises(RegistryError): + self.registry.render("find_object", {}) + + def test_unknown_skill_is_rejected(self) -> None: + with self.assertRaises(RegistryError): + self.registry.render("not_registered", {}) + + def test_stop_hold_requires_explicit_resume(self) -> None: + decision = self.controller.process(self.event("鍋滀竴涓", 1, 1)) + self.assertEqual(decision.action, "stop_speech") + self.assertTrue(self.controller.state.speech_hold_active) + self.assertTrue(self.controller.state.drop_output_until_listen) + self.controller.mark_listen_fence() + self.assertTrue(self.controller.state.speech_hold_active) + self.assertTrue(self.controller.state.drop_output_until_listen) + resumed = self.controller.process(self.event("鎭㈠瀵硅瘽", 2, 2)) + self.assertEqual(resumed.action, "resume_speech") + self.assertFalse(self.controller.state.speech_hold_active) + self.assertFalse(self.controller.state.drop_output_until_listen) + + def test_reset_always_supplies_idle_prompt(self) -> None: + decision = self.controller.process(self.event("閲嶆柊寮濮", 1, 1)) + self.assertEqual(decision.action, "restart_session") + self.assertIn("AI 鐪奸暅鍔╂墜", decision.payload["system_prompt"]) + self.assertTrue(decision.payload["prompt_path"].endswith("idle_chat_zh.txt")) + + def test_skill_activation_supplies_rendered_prompt_path_and_hash(self) -> None: + decision = self.controller.process(self.event("甯垜鎵炬墜鏈", 1, 1)) + self.assertEqual(decision.action, "activate_skill") + self.assertIn("褰撳墠瀵绘壘鐩爣锛氭墜鏈", decision.payload["system_prompt"]) + self.assertTrue(decision.payload["prompt_path"].endswith("find_object_zh.txt")) + self.assertEqual(len(decision.payload["prompt_sha256"]), 64) + + def test_duplicate_asr_event_is_ignored(self) -> None: + first = self.controller.process(self.event("鍋滀竴涓", 1, 9)) + second = self.controller.process(self.event("鍋滀竴涓", 2, 9)) + self.assertTrue(first.accepted) + self.assertEqual(second.reason, "duplicate_asr_event") + + def test_same_skill_same_slots_does_not_restart(self) -> None: + first = self.controller.process(self.event("甯垜鎵炬墜鏈", 1, 1)) + self.assertTrue(first.accepted) + self.controller.mark_restart_complete("find_object", {"target": "鎵嬫満"}, 1) + second = self.controller.process(self.event("甯垜鎵炬墜鏈", 2, 2), now_ms=9000) + self.assertFalse(second.accepted) + self.assertEqual(second.reason, "same_skill_same_slots") + + def test_restart_pending_is_last_write_wins(self) -> None: + self.controller.process(self.event("甯垜鎵炬墜鏈", 1, 1)) + queued_one = self.controller.process(self.event("甯垜鎵鹃挜鍖", 2, 2)) + queued_two = self.controller.process(self.event("璇讳竴涓", 3, 3)) + self.assertEqual(queued_one.action, "queue_skill") + self.assertEqual(queued_two.action, "queue_skill") + pending = self.controller.mark_restart_complete("find_object", {"target": "鎵嬫満"}, 1) + self.assertEqual(pending, {"skill_id": "read_text", "slots": {}}) + + def test_disconnect_clears_restart_and_pending(self) -> None: + self.controller.process(self.event("甯垜鎵炬墜鏈", 1, 1)) + self.controller.process(self.event("璇讳竴涓", 2, 2)) + self.controller.mark_disconnected() + self.assertFalse(self.controller.state.restart_in_progress) + self.assertIsNone(self.controller.state.pending_skill) + + +class EchoAndVADTests(unittest.TestCase): + def test_recent_model_echo_is_blocked_but_stop_is_allowed(self) -> None: + guard = EchoGuard(window_ms=20_000, similarity_threshold=0.8) + guard.note_model_text("璇峰憡璇夋垜杩橀渶瑕佷粈涔堝府鍔", at_ms=1000) + blocked = guard.evaluate("璇峰憡璇夋垜杩橀渶瑕佷粈涔堝府鍔", ControlIntent.NONE, at_ms=1100) + self.assertFalse(blocked.allow) + guard.set_ai_speaking(True) + allowed = guard.evaluate("鍋滀竴涓", ControlIntent.STOP_SPEECH, at_ms=1200) + self.assertTrue(allowed.allow) + resumed = guard.evaluate("鎭㈠瀵硅瘽", ControlIntent.RESUME_SPEECH, at_ms=1300) + self.assertTrue(resumed.allow) + + skill = guard.evaluate("甯垜鎵句竴涓嬫垜鐨勬墜鏈", ControlIntent.ACTIVATE_SKILL, at_ms=1400) + self.assertTrue(skill.allow) + returned = guard.evaluate("鍥炲埌鏅氳亰澶", ControlIntent.RETURN_TO_CHAT, at_ms=1450) + self.assertTrue(returned.allow) + cancelled = guard.evaluate("鍙栨秷浠诲姟", ControlIntent.CANCEL_SKILL, at_ms=1475) + self.assertTrue(cancelled.allow) + ordinary = guard.evaluate("浠婂ぉ澶╂皵鎬庝箞鏍", ControlIntent.NONE, at_ms=1500) + self.assertFalse(ordinary.allow) + self.assertEqual(ordinary.reason, "ordinary_skill_suppressed_while_ai_speaking") + + def test_vad_emits_one_utterance(self) -> None: + vad = EnergyVAD(rms_threshold=0.01, min_speech_ms=100, end_silence_ms=200) + frames = [np.zeros(1600, np.float32)] + frames += [np.full(1600, 0.1, np.float32) for _ in range(3)] + frames += [np.zeros(1600, np.float32) for _ in range(3)] + utterances = [] + for index, frame in enumerate(frames): + result = vad.feed(frame, index * 100.0) + if result is not None: + utterances.append(result) + self.assertEqual(len(utterances), 1) + self.assertGreater(utterances[0].audio.size, 0) + + +class ModelTurnLogTests(unittest.TestCase): + def test_streamed_fragments_are_aggregated_with_audio_duration(self) -> None: + turns = ModelTurnAccumulator() + first = turns.feed( + { + "state": "speak", + "session_id": "s1", + "generation": 2, + "skill_id": "read_text", + "text": "涓婃捣", + "audio_ms": 120, + } + ) + self.assertEqual(first, []) + second = turns.feed( + { + "state": "speak", + "session_id": "s1", + "generation": 2, + "skill_id": "read_text", + "text": "鐢靛姏", + "audio_ms": 180, + "decode_end": True, + } + ) + self.assertEqual(second, []) + completed = turns.feed( + { + "state": "listen", + "session_id": "s1", + "generation": 2, + "skill_id": "read_text", + "end_of_turn": True, + } + ) + self.assertEqual(completed[0]["text"], "涓婃捣鐢靛姏") + self.assertEqual(completed[0]["audio_ms"], 300.0) + + +class CVIsolationTests(unittest.IsolatedAsyncioTestCase): + async def test_shadow_provider_contains_failures(self) -> None: + class Broken: + def analyze(self, *args, **kwargs): + raise RuntimeError("synthetic CV failure") + + result = ShadowCVProvider(Broken(), "broken").analyze( + None, "f1", 1.0, "find_object", {} + ) + self.assertEqual(result.provider, "shadow:broken") + self.assertIn("synthetic CV failure", result.error or "") + + async def test_pipeline_is_non_blocking_and_latest_frame_wins(self) -> None: + observations: list[CVObservation] = [] + + class Slow: + def analyze(self, frame, frame_id, timestamp_ms, skill_id, slots): + time.sleep(0.08) + return CVObservation( + frame_id=frame_id, + timestamp_ms=timestamp_ms, + skill_id=skill_id, + provider="slow", + ) + + pipeline = CVPipeline( + CVProviderRegistry({"slow": Slow()}), + on_observation=observations.append, + inference_timeout_ms=500, + ) + pipeline.submit(FrameEnvelope(b"1", "f1", 1.0, "find_object", {}, "shadow", "slow")) + # submit() must return before the synchronous provider completes. A + # wall-clock threshold is flaky on a loaded Windows workstation. + self.assertEqual(observations, []) + self.assertIsNotNone(pipeline.worker_task) + await asyncio.sleep(0.01) + pipeline.submit(FrameEnvelope(b"2", "f2", 2.0, "find_object", {}, "shadow", "slow")) + pipeline.submit(FrameEnvelope(b"3", "f3", 3.0, "find_object", {}, "shadow", "slow")) + for _ in range(100): + if len(observations) >= 2: + break + await asyncio.sleep(0.01) + snapshot = pipeline.snapshot() + await pipeline.close() + self.assertEqual([item.frame_id for item in observations], ["f1", "f3"]) + self.assertEqual(snapshot["dropped_frames"], 1) + + async def test_pipeline_timeout_is_observed_without_escaping(self) -> None: + observations: list[CVObservation] = [] + + class TooSlow: + def analyze(self, frame, frame_id, timestamp_ms, skill_id, slots): + time.sleep(0.08) + return CVObservation(frame_id, timestamp_ms, skill_id, "too_slow") + + pipeline = CVPipeline( + CVProviderRegistry({"too_slow": TooSlow()}), + on_observation=observations.append, + inference_timeout_ms=10, + ) + pipeline.submit( + FrameEnvelope(b"x", "timeout", 1.0, "find_object", {}, "shadow", "too_slow") + ) + for _ in range(50): + if observations: + break + await asyncio.sleep(0.005) + snapshot = pipeline.snapshot() + await pipeline.close() + self.assertIn("TimeoutError", observations[0].error or "") + self.assertEqual(snapshot["timeout_count"], 1) + + +if __name__ == "__main__": + unittest.main() diff --git a/extensions/assistive_harness/tests/test_cv_yolo.py b/extensions/assistive_harness/tests/test_cv_yolo.py new file mode 100644 index 0000000..59476f8 --- /dev/null +++ b/extensions/assistive_harness/tests/test_cv_yolo.py @@ -0,0 +1,41 @@ +from __future__ import annotations + +import unittest +from pathlib import Path + +import cv2 +import numpy as np + +from extensions.assistive_harness.cv.yolo_onnx import YoloOnnxProvider + + +MODEL = ( + Path(__file__).resolve().parents[4] + / "OmniHarness" + / "mini_omni_harness" + / "models" + / "yolo26n.onnx" +) + + +@unittest.skipUnless(MODEL.is_file(), "local YOLO reference weights are unavailable") +class YoloOnnxSmokeTests(unittest.TestCase): + def test_blank_jpeg_runs_real_onnx_inference(self) -> None: + ok, encoded = cv2.imencode(".jpg", np.zeros((480, 640, 3), dtype=np.uint8)) + self.assertTrue(ok) + observation = YoloOnnxProvider(MODEL).analyze( + encoded.tobytes(), + "blank", + 1.0, + "find_object", + {"target": "鎵嬫満"}, + ) + self.assertEqual(observation.provider, "yolo_onnx") + self.assertEqual(observation.values["status"], "ok") + self.assertEqual(observation.values["canonical_label"], "cell phone") + self.assertEqual(observation.values["model"], "yolo26n.onnx") + self.assertIsInstance(observation.values["detections"], list) + + +if __name__ == "__main__": + unittest.main() diff --git a/extensions/assistive_harness/tests/test_phase_b_rokid.py b/extensions/assistive_harness/tests/test_phase_b_rokid.py new file mode 100644 index 0000000..1787b69 --- /dev/null +++ b/extensions/assistive_harness/tests/test_phase_b_rokid.py @@ -0,0 +1,352 @@ +from __future__ import annotations + +import asyncio +import base64 +import unittest +from pathlib import Path +from typing import Any + +import numpy as np + +from extensions.assistive_harness.phase_b.rokid_runtime import ( + AudioMirrorChunker, + DropOldestAudioQueue, + GatewaySessionManager, + LatestFrame, + PCSpeaker, + PhaseBRokidRuntime, + RokidRuntimeConfig, + SessionSpec, + apply_pcm16_gain, + make_session_ready_chime, + pcm16le_to_float32, +) +from extensions.assistive_harness.registry import SkillRegistry + + +CONFIG = Path(__file__).resolve().parents[1] / "config" / "skills.example.yaml" + + +class FakeSpeaker: + def __init__(self) -> None: + self.blocked = False + self.flush_count = 0 + self.resume_count = 0 + self.enqueued: list[tuple[np.ndarray, int]] = [] + + async def start(self) -> None: + pass + + async def enqueue(self, samples: np.ndarray, generation: int) -> None: + if not self.blocked: + self.enqueued.append((samples.copy(), generation)) + + async def block_and_flush(self) -> None: + self.blocked = True + self.flush_count += 1 + + async def resume(self) -> None: + self.blocked = False + self.resume_count += 1 + + async def close(self) -> None: + pass + + def pending_ms(self) -> float: + return 0.0 + + +class FakeTelemetry: + connected = True + + def __init__(self) -> None: + self.messages: list[dict[str, Any]] = [] + + async def send(self, payload: dict[str, Any]) -> None: + self.messages.append(dict(payload)) + + +class FakeSession: + next_id = 0 + + def __init__( + self, + _config: RokidRuntimeConfig, + spec: SessionSpec, + _audio_queue: DropOldestAudioQueue, + _latest_frame: LatestFrame, + _gate: object, + on_result: object, + ) -> None: + type(self).next_id += 1 + self.spec = spec + self.on_result = on_result + self.session_id = f"fake-{type(self).next_id}" + self.status = "created" + self.last_error = "" + self.started = False + self.stopped_with: str | None = None + self.injected_tasks: list[str] = [] + + async def start(self) -> None: + self.started = True + self.status = "running" + + async def stop(self, cleanup_mode: str) -> None: + self.stopped_with = cleanup_mode + self.status = "stopped" + + async def inject_task(self, text: str) -> bool: + self.injected_tasks.append(text) + return True + + +class RokidAudioBoundaryTests(unittest.IsolatedAsyncioTestCase): + def test_health_marks_device_input_not_ready_before_first_packet(self) -> None: + runtime = PhaseBRokidRuntime( + RokidRuntimeConfig(skills_config=str(CONFIG), play_audio=False), + speaker=FakeSpeaker(), + session_factory=FakeSession, + ) + self.assertFalse(runtime.health()["device_input_ready"]) + runtime.stats.audio_packets = 1 + self.assertTrue(runtime.health()["device_input_ready"]) + + def test_pcm16_conversion_and_mirror_chunking(self) -> None: + raw = np.array([-32768, 0, 32767], dtype=" None: + raw = np.array([-4_000, 1_000, 4_000], dtype=" None: + chime = make_session_ready_chime() + self.assertEqual(chime.dtype, np.float32) + self.assertGreater(chime.size, 24_000 * 0.15) + self.assertLess(chime.size, 24_000 * 0.25) + self.assertLessEqual(float(np.max(np.abs(chime))), 0.321) + self.assertGreater(float(np.max(np.abs(chime))), 0.30) + + async def test_drop_oldest_queue_preserves_newest_packets(self) -> None: + queue = DropOldestAudioQueue(max_packets=2) + queue.put_nowait(b"old") + queue.put_nowait(b"middle") + queue.put_nowait(b"new") + self.assertEqual(queue.dropped_packets, 1) + self.assertEqual(await queue.get(0.01), b"middle") + self.assertEqual(await queue.get(0.01), b"new") + + async def test_pc_speaker_stop_closes_and_resume_reopens_stream(self) -> None: + class FakeOutputStream: + def __init__(self) -> None: + self.abort_count = 0 + self.close_count = 0 + + def abort(self) -> None: + self.abort_count += 1 + + def close(self) -> None: + self.close_count += 1 + + speaker = PCSpeaker() + old_stream = FakeOutputStream() + new_stream = FakeOutputStream() + speaker._stream = old_stream + + await speaker.block_and_flush() + self.assertTrue(speaker.blocked) + self.assertIsNone(speaker._stream) + self.assertEqual(old_stream.abort_count, 1) + self.assertEqual(old_stream.close_count, 1) + + speaker._open_stream = lambda: setattr(speaker, "_stream", new_stream) + await speaker.resume() + self.assertFalse(speaker.blocked) + self.assertIs(speaker._stream, new_stream) + + +class PhaseBControlContractTests(unittest.IsolatedAsyncioTestCase): + def setUp(self) -> None: + FakeSession.next_id = 0 + self.registry = SkillRegistry(CONFIG) + self.speaker = FakeSpeaker() + self.telemetry = FakeTelemetry() + self.sessions: list[FakeSession] = [] + + def factory(*args: Any) -> FakeSession: + session = FakeSession(*args) + self.sessions.append(session) + return session + + self.manager = GatewaySessionManager( + RokidRuntimeConfig( + skills_config=str(CONFIG), + play_audio=False, + playback_echo_tail_s=0.0, + ), + self.registry, + DropOldestAudioQueue(max_packets=4), + LatestFrame(), + self.speaker, + session_factory=factory, + ) + self.manager.harness = self.telemetry + + async def test_stop_and_resume_flush_pc_output_and_ack(self) -> None: + await self.manager.start_initial() + stop_ack = await self.manager.handle_control( + { + "type": "control.intent", + "event_id": 1, + "intent": "stop_speech", + "accepted": True, + } + ) + self.assertTrue(stop_ack["ok"]) + self.assertTrue(self.manager.gate.speech_hold_active) + self.assertTrue(self.speaker.blocked) + self.assertEqual(self.speaker.flush_count, 1) + + resume_ack = await self.manager.handle_control( + { + "type": "control.intent", + "event_id": 2, + "intent": "resume_speech", + "accepted": True, + } + ) + self.assertTrue(resume_ack["ok"]) + self.assertFalse(self.manager.gate.speech_hold_active) + self.assertFalse(self.speaker.blocked) + self.assertEqual( + [message["intent"] for message in self.telemetry.messages if message.get("type") == "control.ack"], + ["stop_speech", "resume_speech"], + ) + + async def test_skill_switch_replaces_session_and_fences_old_output(self) -> None: + await self.manager.start_initial() + old_session = self.sessions[0] + ack = await self.manager.handle_control( + { + "type": "control.intent", + "event_id": 3, + "intent": "activate_skill", + "accepted": True, + "skill_id": "read_text", + "slots": {}, + } + ) + new_session = self.sessions[1] + self.assertTrue(ack["ok"]) + self.assertEqual(old_session.stopped_with, "light") + self.assertTrue(new_session.started) + self.assertNotEqual(old_session.session_id, new_session.session_id) + self.assertEqual(self.manager.gate.generation, 1) + self.assertEqual(self.manager.gate.current_skill, "read_text") + self.assertEqual( + new_session.injected_tasks, + ["璇风珛鍗宠鍙栧綋鍓嶇敾闈腑鏈鏄庢樉鐨勬枃瀛楋紝鍙鐪嬪埌鐨勫唴瀹广"], + ) + self.assertTrue(ack["task_trigger_sent"]) + self.assertFalse(self.speaker.blocked) + self.assertTrue( + any( + message.get("type") == "session.state" + and message.get("phase") == "restart_complete" + for message in self.telemetry.messages + ) + ) + + audio = base64.b64encode(np.ones(10, dtype=np.float32).tobytes()).decode() + await self.manager.handle_result( + old_session, + {"type": "result", "text": "stale", "audio_data": audio}, + ) + self.assertEqual(self.manager.gate.dropped_old_text, 1) + self.assertEqual(self.manager.gate.dropped_old_audio, 1) + self.assertEqual(self.speaker.enqueued, []) + + await self.manager.handle_result( + new_session, + {"type": "result", "text": "current", "audio_data": audio}, + ) + self.assertEqual(len(self.speaker.enqueued), 1) + self.assertEqual(self.speaker.enqueued[0][1], 1) + + async def test_restart_complete_plays_one_local_ready_chime(self) -> None: + self.manager.config.play_audio = True + await self.manager.start_initial() + self.assertEqual(self.speaker.enqueued, []) + + ack = await self.manager.handle_control( + { + "type": "control.intent", + "event_id": 4, + "intent": "reset_session", + "accepted": True, + "skill_id": "idle_chat", + "slots": {}, + } + ) + + self.assertTrue(ack["ok"]) + self.assertEqual(len(self.speaker.enqueued), 1) + cue, generation = self.speaker.enqueued[0] + self.assertEqual(generation, 1) + self.assertGreater(cue.size, 0) + self.assertGreater(self.manager.gate.local_cue_mute_until_mono, 0.0) + + async def test_obstacle_switch_injects_one_shot_visual_task(self) -> None: + await self.manager.start_initial() + ack = await self.manager.handle_control( + { + "type": "control.intent", + "event_id": 5, + "intent": "activate_skill", + "accepted": True, + "skill_id": "obstacle_avoidance", + "slots": {}, + } + ) + self.assertTrue(ack["task_trigger_sent"]) + self.assertIn("鍒ゆ柇褰撳墠鐢婚潰", self.sessions[1].injected_tasks[0]) + + async def test_model_log_turn_ends_on_listen_not_decode_slice(self) -> None: + await self.manager.start_initial() + session = self.sessions[0] + await self.manager.handle_result( + session, + {"type": "result", "text": "涓婃捣", "end_of_turn": True}, + ) + await self.manager.handle_result( + session, + {"type": "result", "text": "鐢靛姏", "end_of_turn": True}, + ) + await self.manager.handle_result( + session, + {"type": "result", "is_listen": True}, + ) + states = [ + item for item in self.telemetry.messages + if item.get("type") == "model.state" + ] + self.assertEqual([item["end_of_turn"] for item in states], [False, False, True]) + self.assertEqual([item["decode_end"] for item in states], [True, True, False]) + + +if __name__ == "__main__": + unittest.main() diff --git a/runtime/openglass_omni/devices.example.json b/runtime/openglass_omni/devices.example.json new file mode 100644 index 0000000..5cecb6c --- /dev/null +++ b/runtime/openglass_omni/devices.example.json @@ -0,0 +1,7 @@ +锘縶 + "_璇存槑": "澶氬壇鐪奸暅閰嶇疆銆傜儳褰曞悗浠庝覆鍙g湅鍒 IP锛屾敼杩欓噷瀵瑰簲閭e壇鐨 esp32_host銆俽otate 鏄鐪奸暅鎽勫儚澶撮『鏃堕拡鏃嬭浆瑙掑害(0/90/180/270)锛屾寜璁惧渚ц鏂瑰悜濉備袱鍙扮瑪璁版湰鍚勮嚜缁存姢鑷繁杩欎唤鏂囦欢銆俫ateway 璧版湰鏈洪粯璁わ紝涓嶆斁杩欓噷銆", + "devices": [ + { "name": "1鍙烽暅", "esp32_host": "esp32_ip", "esp32_port": 80, "rotate": 0 }, + { "name": "2鍙烽暅", "esp32_host": "esp32_ip2", "esp32_port": 80, "rotate": 90 } + ] +} \ No newline at end of file diff --git a/runtime/openglass_omni/panel.html b/runtime/openglass_omni/panel.html index 99e4173..c024489 100644 --- a/runtime/openglass_omni/panel.html +++ b/runtime/openglass_omni/panel.html @@ -107,6 +107,10 @@ + + @@ -184,11 +188,20 @@

绗竴瑙嗚

(CFG.devices||[]).forEach(d=>{ const o=document.createElement("option"); o.value=d; o.textContent=d; sel.appendChild(o); }); - applyChain(); // 鎸夐摼璺埛鏂扮伅/鏃ュ織椤电/FPV/鐪奸暅涓嬫媺鍙鎬 + // 濉厖鍒ゆ嵁涓嬫媺锛堝彧鏈夊紑浜嗘紡鏂楃殑閾捐矾鐢ㄥ緱涓婏紝applyChain 閲屾帶鍒舵樉闅愶級 + const ssel = document.getElementById("sceneSel"); + (CFG.scenes||[]).forEach(x=>{ + const o=document.createElement("option"); o.value=x; o.textContent=x; ssel.appendChild(o); + }); + if(CFG.default_scene) ssel.value = CFG.default_scene; + applyChain(); // 鎸夐摼璺埛鏂扮伅/鏃ュ織椤电/FPV/鐪奸暅+鍒ゆ嵁涓嬫媺鍙鎬 fpvOn=false; toggleFpv(); // 榛樿寮鍚涓瑙嗚鐢婚潰 setInterval(poll, 1000); } +// 杩涚▼鍐呴儴鍚 -> 鐣岄潰鏄剧ず鍚嶏紙鎷夸笉鍒板氨閫鍥炲唴閮ㄥ悕锛 +function PLABEL(n){ return (CFG.proc_labels && CFG.proc_labels[n]) || n; } + // 鍒囬摼璺細閲嶇敾鐘舵佺伅銆佹棩蹇楅〉绛俱丗PV 鍦板潃锛汻okid 鍒嗘敮闅愯棌鐪奸暅涓嬫媺 function applyChain(){ const c = CFG.chains[curChain]; @@ -197,7 +210,7 @@

绗竴瑙嗚

lights.innerHTML = ""; c.procs.forEach(n=>{ const d=document.createElement("div"); d.className="light"; - d.innerHTML = `${n}`; + d.innerHTML = `${PLABEL(n)}`; lights.appendChild(d); }); // 鏃ュ織椤电锛堜繚鐣 autoscroll 鍕鹃夋锛 @@ -206,7 +219,7 @@

绗竴瑙嗚

tabs.innerHTML = ""; c.procs.forEach((n,i)=>{ const t=document.createElement("div"); - t.className="tab"+(i===0?" active":""); t.dataset.t=n; t.textContent=n; + t.className="tab"+(i===0?" active":""); t.dataset.t=n; t.textContent=PLABEL(n); t.onclick=()=>switchTab(n); tabs.appendChild(t); }); tabs.appendChild(auto); @@ -215,6 +228,10 @@

绗竴瑙嗚

const show = c.need_device; document.getElementById("deviceSel").style.display = show ? "" : "none"; document.getElementById("devLabel").style.display = show ? "" : "none"; + // 鍒ゆ嵁涓嬫媺锛氬彧鏈夊紑浜嗘紡鏂楃殑閾捐矾(鈶⑩懀)鎵嶆樉绀猴紱鈶犫憽娌℃湁婕忔枟锛岄変簡涔熶笉璧蜂綔鐢 + const showScene = !!c.need_scene; + document.getElementById("sceneSel").style.display = showScene ? "" : "none"; + document.getElementById("sceneLabel").style.display = showScene ? "" : "none"; // FPV 鍦板潃闅忛摼璺彉锛圗SP32=demo bridge_ui:8080锛汻okid=bridge:18080 鑷甫椤碉級 document.getElementById("fpvUrl").textContent = c.fpv_url; if(fpvOn){ fpvOn=false; toggleFpv(); } // 閲嶆寕 iframe 鍒版柊鍦板潃 @@ -226,6 +243,8 @@

绗竴瑙嗚

} function curDevice(){ return document.getElementById("deviceSel").value; } function onDeviceChange(){ window.pywebview.api.set_device(curDevice()); } +function curScene(){ const e=document.getElementById("sceneSel"); return e ? e.value : ""; } +function onSceneChange(){ window.pywebview.api.set_scene(curScene()); } function selectPreset(name,el){ document.querySelectorAll(".preset").forEach(p=>p.classList.remove("active")); el.classList.add("active"); @@ -234,7 +253,7 @@

绗竴瑙嗚

} function startAll(){ const p = document.getElementById("prompt").value.trim(); - window.pywebview.api.start_all(p, curDevice(), curChain); + window.pywebview.api.start_all(p, curDevice(), curChain, curScene()); } function stopDemo(){ window.pywebview.api.stop_demo(); } // 鍋滄锛氬彧鍋 demo锛屾棤闇纭锛堢幇鍦鸿蹇級 function restartDemo(){ diff --git a/runtime/openglass_omni/panel.py b/runtime/openglass_omni/panel.py index ed262e3..e16bcb6 100644 --- a/runtime/openglass_omni/panel.py +++ b/runtime/openglass_omni/panel.py @@ -169,7 +169,44 @@ "--connect-retry", ], - # 鈶 rokid bridge 鈥斺 涓 demo 骞宠銆佷簩閫変竴鐨勨滅鍥涗釜杩涚▼鈥濄 + # 鈶 8021 harness 鈥斺 璇煶鎺у埗鑵匡紙鍋滀竴涓/閲嶆柊寮濮/鎵剧墿锛夛紝涔熸槸 ASR 鏉ユ簮銆 + # 鍙湁 鈶♀憿鈶 甯 harness 鐨勯摼璺墠璧峰畠銆 + # 鈽 寮婧愮敤鎴峰繀鏀癸細--model-path 鎸囧悜浣犳満鍣ㄤ笂鐨 FunASR 娴佸紡妯″瀷鐩綍 鈽 + # 璇佷功锛氳嚜绛惧嵆鍙紝鍙负杩 wss 鎻℃墜锛屼笉缁戞満鍣紱瀹㈡埛绔叏鏄 CERT_NONE 涓嶆牎楠屻 + # 鈽 瀹冨湪 extensions/ 涓嬶紝鐢 -m 鍚姩锛屽洜姝 cwd 蹇呴』鏄 MiniCPM-o-Demo 鐩綍 + # 锛堣 _spawn 閲岀殑 name in (...) 鐧藉悕鍗曪級銆 + "harness": [ + "python", "-m", "extensions.assistive_harness.server", "--enabled", + #"--model-path", r"\LocalASRmodel\speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-online", + # e.g. + "--model-path", (r"D:\OpenGlass\OmniDeployment\OpenGlass\LocalASRmodel" + r"\speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-online"), + "--port", "8021", + # 缁濆璺緞锛歨arness 鐨 cwd 鏄 OpenGlass 浠撳簱鏍癸紝鑰岃瘉涔︾敱 _ensure_certs + # 鐢熸垚鍦 minicpm_demo_dir/certs 涓嬶紙gateway 涔熺敤鍚屼竴瀵癸級銆 + "--certfile", "{certfile}", + "--keyfile", "{keyfile}", + ], + + # 鈶 demo_funnel 鈥斺 esp32_runtime锛坔arness + CV 婕忔枟锛夈備笌 鈶 demo 浜掓枼锛 + # 涓よ呴兘瑕佺嫭鍗 ESP32 鐨勯煶棰 WS 鍜屽浘鍍 TCP锛屼笉鑳藉悓鏃惰窇銆 + # 杩欓噷鍙斁鍚勬。浣嶅叡鐢ㄧ殑鍙傛暟锛--esp32-host/--esp32-port/--rotate 鐢 + # devices.json 濉紙_device_args锛夛紝婕忔枟妗d綅鍙傛暟鐢 _funnel_extra_args 杩藉姞銆 + # 鈽 鍚屾牱鍦 extensions/ 涓嬶紝cwd 蹇呴』鏄 MiniCPM-o-Demo 鐩綍銆 + # 鈽 8006 鏄 TLS锛坔ttps:// 鑳藉彇鍒 openapi.json锛宧ttp:// 鏄 Empty reply锛夛紝 + # 璧伴粯璁 wss锛**涓嶈**鍔 --no-tls锛涜 harness 8021 鏄嚜绛 wss锛屼袱鑰呴兘鏄 wss + # 浣嗚瘉涔︽潵婧愪笉鍚屻 + "demo_funnel": [ + "python", "-m", "extensions.assistive_harness.phase_b.esp32_runtime", + "--gateway", "localhost:8006", + "--harness-url", "wss://127.0.0.1:8021/ws/control", + "--image-tcp-port", "5000", + "--web-ui-port", "8080", + "--record-live", + "--prompt", "{prompt}", + ], + + # 鈶 rokid bridge 鈥斺 涓 demo 骞宠銆佷簩閫変竴鐨勨滅鍥涗釜杩涚▼鈥濄 # 娉ㄦ剰锛氬畠涓嶆槸瀹㈡埛绔幓杩炵溂闀滐紝鑰屾槸鍦 PC 涓婂紑 18080 绔彛绛 APK 杩炶繘鏉ャ # 鈽 鐢 v8锛圓PI V2锛夈倂7 鏄棫鍗忚(/ws/duplex)锛岃繛涓嶄笂鏂 gateway:8006銆 # 鈽 run_rokid.ps1 / run_rokid_wifi.cmd 宸蹭笉鍐嶉渶瑕佲斺斿畠浠仛鐨勪簨 @@ -191,8 +228,12 @@ # live.html 瑙傛祴椤碉紙涓 ESP32 demo 鍚屼竴濂 bridge_ui 鍓嶇/妯℃澘锛 "--ui-port", "8080", "--prompt", "{prompt}", - "--glasses-ssid", "SQZ", - "--glasses-psk", "sqz.ac.cn", + # 鈽 鐪奸暅杩炵殑 WiFi銆**涓嶈鎶婄湡瀹炲瘑鐮佹彁浜よ繘浠撳簱** 鈥斺 + # 鍦 panel.py 鍚岀洰褰曟斁涓涓 panel.local.json锛堝凡鍦 .gitignore锛夛細 + # { "glasses_ssid": "浣犵殑WiFi", "glasses_psk": "浣犵殑瀵嗙爜" } + # 娌℃湁璇ユ枃浠舵椂鐢ㄤ笅闈㈢殑鍗犱綅鍊硷紝Rokid 閾捐矾浼氳繛涓嶄笂 WiFi 浣嗗叾浣欏姛鑳芥甯搞 + "--glasses-ssid", "{glasses_ssid}", + "--glasses-psk", "{glasses_psk}", ], }, @@ -202,13 +243,48 @@ # rokid 鈫 rokid_minicpm_v7.py 锛圥C 寮绔彛绛 APK 杩炶繘鏉ワ紝鏃 device锛 "chains": { "esp32": { - "label": "ESP32 鐪奸暅", + "label": "鈶 鍩虹瀵硅瘽", "tail": "demo", # 绗洓绾ц繘绋嬪悕 "start_order": ["llama", "worker", "gateway", "demo"], "stop_order": ["demo", "gateway", "worker", "llama"], "need_device": True, # 鏄剧ず鐪奸暅涓嬫媺 "fpv_key": "fpv_url", # 绗竴瑙嗚鍦板潃 }, + # 鈹鈹 浠ヤ笅涓夋潯璧 esp32_runtime锛坔arness + 婕忔枟锛夛紝鍥涙。閫掕繘婕旂ず 鈹鈹 + # 鈶 鍩虹瀵硅瘽 = 涓婇潰鐨 "esp32"锛坋sp32_bridge.py锛屾棤 harness 鏃犳紡鏂楋級 + # 鈶 璇煶鎺у埗 = harness 寮銆佹紡鏂楀叧 + # 鈶 璐ㄩ噺绛涢 = 鈶 + 姣忕澶氬抚閲屾寫鏈娓呮櫚鐨勪竴寮 + # 鈶 瀹屾暣闃插够瑙 = 鈶 + 鍧忓浘鐩存帴鎷︿笅骞惰闊虫彁绀 + # 姣忔。鍙瘮涓婁竴妗e涓浠朵簨锛氱湅鍒颁粈涔堣浠涔 鈫 鑳藉惉鎳傛寚浠 鈫 鍥句細鎸 鈫 鍧忓浘浼氭嫤 + # 锛堟病鏈"鍙鐒"杩欎竴妗o細瀵圭劍鍚庤绛 settle 鍐嶉噸鎶擄紝1s chunk 鏃跺簭鍥哄畾锛 + # 瀵圭劍鍚庨偅寮犳湭蹇呰刀寰椾笂杩欎竴杞紝绛変簬鑺变簡鏃堕棿娌$敤涓娿傦級 + "esp32_voice": { + "label": "鈶 璇煶鎺у埗", + "tail": "demo_funnel", + "funnel": 0, # 婕忔枟妗d綅锛0 鍏 / 2 閫夊浘 / 3 閫夊浘+鎷掔粷 + "start_order": ["llama", "worker", "gateway", "harness", "demo_funnel"], + "stop_order": ["demo_funnel", "harness", "gateway", "worker", "llama"], + "need_device": True, + "fpv_key": "fpv_url", + }, + "esp32_select": { + "label": "鈶 璐ㄩ噺绛涢", + "tail": "demo_funnel", + "funnel": 2, + "start_order": ["llama", "worker", "gateway", "harness", "demo_funnel"], + "stop_order": ["demo_funnel", "harness", "gateway", "worker", "llama"], + "need_device": True, + "fpv_key": "fpv_url", + }, + "esp32_full": { + "label": "鈶 瀹屾暣闃插够瑙", + "tail": "demo_funnel", + "funnel": 3, + "start_order": ["llama", "worker", "gateway", "harness", "demo_funnel"], + "stop_order": ["demo_funnel", "harness", "gateway", "worker", "llama"], + "need_device": True, + "fpv_key": "fpv_url", + }, "rokid": { "label": "Rokid 鐪奸暅", "tail": "rokid", @@ -220,6 +296,30 @@ }, "default_chain": "esp32", + # 鍒ゆ嵁妗d綅锛氭紡鏂楃殑涓ゅ鏍囧畾鍙傛暟銆傚澶栫敤鍦烘櫙鍚嶏紝涓嶆毚闇插唴閮ㄥ笺 + # 涓ユ牸 = medicine 锛堣嵂鐩掑皬瀛楅珮鍗憋紝worst_x0=6.5 ACCEPT_Q=0.55锛 + # 鏃ュ父 = stationery锛堢敓娲荤敤鍝侊紝worst_x0=5.0 ACCEPT_Q=0.48锛 + "scenes": {"涓ユ牸鍒ゆ嵁": "medicine", "鏃ュ父鍒ゆ嵁": "stationery"}, + "default_scene": "涓ユ牸鍒ゆ嵁", + + # 妗d綅鈶g殑鎻愮ず闊崇洰褰曘**璺熺潃 extensions 鍖呰蛋**锛堝拰 bridge_ui 鐨 templates/ 鍚屾濊矾锛夛紝 + # 涓嶈惤鍦ㄤ笂娓 MiniCPM-o-Demo 閲岋紝淇濇寔涓変粨搴撶嫭绔嬨傝繖閲屾槸鐩稿 OpenGlass 浠撳簱鏍癸紝 + # panel 浼氭嫾鎴愮粷瀵硅矾寰勪紶缁 esp32_runtime锛堝畠鐨 cwd 鏄笂娓革紝鐩稿璺緞浼氭寚閿欏湴鏂癸級銆 + # 鍚姩鍓嶆鏌ワ紝缂轰簡灏辨姤閿 鈥斺 鑰屼笉鏄窇璧锋潵鎵嶅彂鐜版病澹伴煶锛岄偅鏃 duplex 宸插湪璺戯紝 + # 娌℃硶褰撳満鐢 gen_reject_wavs.py 鐢熸垚銆 + "reject_wav_dir": "extensions/assistive_harness/phase_b/assets/reject_wav", + + # 杩涚▼鐨勭晫闈㈡樉绀哄悕锛堢伅娉/鏃ュ織椤电鐢級銆傚唴閮ㄥ悕涓嶅彉锛屽彧褰卞搷 UI銆 + "proc_labels": { + "llama": "鎺ㄧ悊鍚庣", + "worker": "worker", + "gateway": "gateway", + "harness": "璇煶鎺у埗(8021)", + "demo": "鐪奸暅", + "demo_funnel": "鐪奸暅+婕忔枟", + "rokid": "Rokid", + }, + # 鍙夌溂闀滐紙瀵瑰簲 devices.json 閲岀殑 name锛夈傞潰鏉块《閮ㄤ笅鎷夐夋嫨銆備粎 ESP32 鍒嗘敮鐢ㄣ "devices": ["宸﹂暅", "鍙抽暅", "澶囩敤闀"], @@ -278,6 +378,191 @@ def _load_devices_from_json(path="devices.json"): # ============================================================================ +def _load_device_map(path="devices.json"): + """璇绘暣浠借澶囪〃锛歯ame -> {host, port, rotate}銆 + + 鍘熸潵鍙彇 name锛堜笅鎷夌敤锛夈備絾 esp32_bridge 鑷繁璇 json锛--device-config锛夛紝 + 鑰 esp32_runtime 涓嶈锛屽畠鍙 --esp32-host/--esp32-port/--rotate锛 + 鎵浠ョ敱 panel 鎶婅繖鍑犻」鍙栧嚭鏉ュ~銆俽otate 灏ゅ叾涓嶈兘婕忥細鎽勫儚澶寸墿鐞嗕晶瑁咃紝 + 涓嶈浆姝g殑璇濇紡鏂楃殑鏂瑰悜妫娴嬩細鎶"鐩告満渚ц"璇垽鎴"鐢ㄦ埛鎶婄洅瀛愭嬁鍙嶄簡"銆 + utf-8-sig锛氳浜嬫湰/VSCode 瀛樼殑 json 鍙兘甯 BOM锛岀敤 utf-8 璇讳細鐐稿湪绗竴涓瓧绗︺ + """ + out = {} + try: + with open(path, "r", encoding="utf-8-sig") as f: + data = json.load(f) + for d in data.get("devices", []): + n = d.get("name") + if not n: + continue + out[n] = { + "host": d.get("esp32_host", ""), + "port": int(d.get("esp32_port", 80)), + "rotate": int(d.get("rotate", 0)) % 360, + } + except FileNotFoundError: + print(f"[panel] 鏈壘鍒 {path}锛屽甫婕忔枟鐨勯摼璺(鈶♀憿鈶)灏嗘棤娉曡嚜鍔ㄥ~ IP") + except Exception as e: + print(f"[panel] 璇诲彇 {path} 澶辫触({e})") + return out + + +CONFIG["device_map"] = _load_device_map(_dev_path) +if CONFIG["device_map"]: + print("[panel] 璁惧璇︽儏: " + ", ".join( + f"{k}({v['host']} rot={v['rotate']})" for k, v in CONFIG["device_map"].items())) + + +def _load_runtime_local(): + """璇 runtime.local.json 鈥斺 鏈満璺緞涓庣瀵嗛厤缃紝**涓嶈繘浠撳簱**銆 + + 涓轰粈涔堣瀹冿細panel.py 閲屽師鏈啓姝讳簡浜斿鏈満缁濆璺緞 + 锛坈onda 鐜銆丮iniCPM-o-Demo 鐩綍銆乴lama-omni-server.exe銆佷富 gguf銆丗unASR 妯″瀷锛夛紝 + 杩樻湁鐪奸暅 WiFi 鐨勬槑鏂囧瘑鐮併傚紑婧愬悗姣忎釜浜洪兘瑕佹敼婧愮爜鎵嶈兘璺戯紝瀵嗙爜涔熶細杩 git 鍘嗗彶銆 + 鏀规垚浠庤繖涓枃浠惰锛宑lone 涓嬫潵鍙渶 `cp runtime.example.json runtime.local.json` 鍐嶅~銆 + + 閿悕娌跨敤浠撳簱閲屽凡鏈夌殑 runtime.example.json 椋庢牸銆 + 鎵句笉鍒版枃浠舵椂淇濈暀 CONFIG 閲岀殑榛樿鍊硷紙涔熷氨鏄師鏉ョ殑鍐欐鍊硷級锛岃涓轰笉鍙樸 + """ + here = os.path.dirname(os.path.abspath(__file__)) + f = os.path.join(here, "runtime.local.json") + if not os.path.isfile(f): + print(f"[panel] 鏈壘鍒 {f}") + print("[panel] 璇峰鍒 runtime.example.json 涓 runtime.local.json 骞跺~鍐欐湰鏈鸿矾寰勶紱") + print("[panel] 鍚﹀垯灏嗘部鐢 panel.py 閲岀殑榛樿鍊硷紙澶氬崐涓嶆槸浣犵殑璺緞锛夈") + return {} + try: + with open(f, "r", encoding="utf-8-sig") as fh: + cfg = json.load(fh) + except Exception as e: + print(f"[panel] 璇诲彇 runtime.local.json 澶辫触({e})锛屾部鐢ㄩ粯璁ゅ") + return {} + + def _expand(v): + return os.path.expandvars(os.path.expanduser(v)) if isinstance(v, str) else v + + n = 0 + # 鈶 绠鍗曢敭 -> CONFIG 椤跺眰 + for src_key, dst_key in (("conda_env", "conda_env"), + ("minicpm_demo_root", "minicpm_demo_dir")): + v = _expand(cfg.get(src_key)) + if v: + CONFIG[dst_key] = v + n += 1 + # 鈶 闇瑕佹浛鎹㈠埌鍛戒护琛屾暟缁勯噷鐨 + def _sub(proc, old_pred, new_val): + """鎶 procs[proc] 閲屾弧瓒 old_pred 鐨勯偅涓椤规崲鎴 new_val銆""" + arr = CONFIG["procs"].get(proc) + if not arr or not new_val: + return 0 + for i, x in enumerate(arr): + if isinstance(x, str) and old_pred(x): + arr[i] = new_val + return 1 + return 0 + + n += _sub("llama", lambda x: x.lower().endswith("llama-omni-server.exe") + or x.lower().endswith("llama-omni-server"), + _expand(cfg.get("llama_server"))) + n += _sub("llama", lambda x: x.lower().endswith(".gguf"), + _expand(cfg.get("llama_model"))) + n += _sub("harness", lambda x: "speech_paraformer" in x or "LocalASRmodel" in x, + _expand(cfg.get("asr_model"))) + print(f"[panel] 宸茶鍏 runtime.local.json锛堢敓鏁 {n} 椤癸級") + return cfg + + +def _load_local_secrets(rt): + """鐪奸暅 WiFi锛圧okid 閾捐矾浼犵粰 APK锛夈傚悓鏍锋潵鑷 runtime.local.json锛 + 鍐欐鍦 panel.py 閲屽氨绛変簬鏄庢枃瀵嗙爜杩涘叕寮浠撳簱銆""" + out = {"glasses_ssid": "", "glasses_psk": ""} + g = (rt or {}).get("glasses") or {} + for k in ("glasses_ssid", "glasses_psk"): + v = g.get(k.replace("glasses_", "")) or (rt or {}).get(k) + if v: + out[k] = v + return out + + +_RTLOCAL = _load_runtime_local() +CONFIG["local"] = _load_local_secrets(_RTLOCAL) + + +def _ensure_certs(base_dir): + """纭繚 certs/cert.pem + key.pem 瀛樺湪锛屾病鏈夊氨鑷涓瀵广 + + gateway(8006) 鍜 harness(8021) 閮界敤杩欎竴瀵癸紙gateway.py 鐨勯粯璁ゅ煎氨鏄 + certs/cert.pem锛岀己浜嗕細鐩存帴鎶ラ敊閫鍑猴級锛屾墍浠ュ洓鏉¢摼璺兘闇瑕佸畠銆 + 鑷璇佷功**涓嶇粦鏈哄櫒**锛岄噷闈㈡病鏈夌‖浠朵俊鎭紱瀹㈡埛绔叏鏄 CERT_NONE 涓嶆牎楠岋紝 + 鍙槸涓轰簡璁 wss 鎻℃墜鑳借繃銆傛墍浠ユ湰鍦扮敓鎴愪竴浠藉嵆鍙紝涓嶅繀涔熶笉璇ユ彁浜よ繘浠撳簱銆 + + 浼樺厛鐢 cryptography 搴擄紱娌¤灏遍鍥炶皟 openssl锛涢兘涓嶈灏辨墦鍑烘墜鍔ㄥ懡浠ゃ + """ + if not base_dir or not os.path.isdir(base_dir): + return False + d = os.path.join(base_dir, "certs") + cert = os.path.join(d, "cert.pem") + key = os.path.join(d, "key.pem") + if os.path.isfile(cert) and os.path.isfile(key): + return True + os.makedirs(d, exist_ok=True) + print(f"[panel] 鏈壘鍒拌瘉涔︼紝姝e湪鐢熸垚鑷璇佷功 -> {d}") + + try: + from cryptography import x509 + from cryptography.x509.oid import NameOID + from cryptography.hazmat.primitives import hashes, serialization + from cryptography.hazmat.primitives.asymmetric import rsa + import datetime as _dt + + k = rsa.generate_private_key(public_exponent=65537, key_size=2048) + name = x509.Name([x509.NameAttribute(NameOID.COMMON_NAME, "localhost")]) + now = _dt.datetime.now(_dt.timezone.utc) + crt = (x509.CertificateBuilder() + .subject_name(name).issuer_name(name) + .public_key(k.public_key()) + .serial_number(x509.random_serial_number()) + .not_valid_before(now - _dt.timedelta(days=1)) + .not_valid_after(now + _dt.timedelta(days=3650)) + .add_extension(x509.SubjectAlternativeName([ + x509.DNSName("localhost"), + x509.IPAddress(__import__("ipaddress").IPv4Address("127.0.0.1")), + ]), critical=False) + .sign(k, hashes.SHA256())) + with open(key, "wb") as f: + f.write(k.private_bytes( + serialization.Encoding.PEM, + serialization.PrivateFormat.TraditionalOpenSSL, + serialization.NoEncryption())) + with open(cert, "wb") as f: + f.write(crt.public_bytes(serialization.Encoding.PEM)) + print("[panel] 璇佷功宸茬敓鎴愶紙cryptography锛屾湁鏁堟湡 10 骞达級") + return True + except ImportError: + pass + except Exception as e: + print(f"[panel] cryptography 鐢熸垚澶辫触({e})锛屾敼璇 openssl") + + try: + subprocess.run( + ["openssl", "req", "-x509", "-newkey", "rsa:2048", + "-keyout", key, "-out", cert, "-days", "3650", + "-nodes", "-subj", "/CN=localhost"], + check=True, capture_output=True, timeout=60) + print("[panel] 璇佷功宸茬敓鎴愶紙openssl锛屾湁鏁堟湡 10 骞达級") + return True + except Exception as e: + print(f"[panel] !! 鑷姩鐢熸垚璇佷功澶辫触: {e}") + print(f"[panel] 璇锋墜鍔ㄦ墽琛岋紙鍦 {base_dir} 涓嬶級锛") + print("[panel] openssl req -x509 -newkey rsa:2048 " + "-keyout certs/key.pem -out certs/cert.pem " + "-days 3650 -nodes -subj \"/CN=localhost\"") + return False + + +_ensure_certs(CONFIG.get("minicpm_demo_dir") or CONFIG.get("cwd")) + + class ProcManager: """绠$悊涓変釜瀛愯繘绋嬶細璧峰仠銆佺姸鎬佽疆璇€佹棩蹇楁敹闆嗐佺鍙/灏辩华鎺㈡祴銆""" @@ -294,6 +579,11 @@ def __init__(self, cfg): self.status = {n: "stopped" for n in cfg["procs"]} # stopped/starting/running/crashed self.current_prompt = next(iter(cfg["presets"].values())) self.current_device = (cfg.get("devices") or ["榛樿"])[0] + self.current_scene = cfg.get("default_scene", "涓ユ牸鍒ゆ嵁") + # 灏捐繘绋嬫槸"涓哄摢濂楅厤缃"璧风殑銆傗憽鈶⑩懀 鍏辩敤 demo_funnel 杩欎竴涓繘绋嬪悕锛 + # 鍙湅杩涚▼娲荤潃灏辫烦杩囧惎鍔ㄧ殑璇濓紝鍒囬摼璺悗鏂扮殑婕忔枟鍙傛暟姘歌繙涓嶄細鐢熸晥 鈥斺 + # UI 鏄剧ず鈶o紝瀹為檯杩樺湪璺戔憽銆傛墍浠ヨ鎸囩汗锛屽彉浜嗗氨閲嶅惎銆 + self._tail_started_for = {} # tail 杩涚▼鍚 -> 鎸囩汗 # 鈥斺 褰撳墠閾捐矾锛歟sp32 / rokid锛屽喅瀹氱鍥涚骇璧峰摢涓繘绋 鈥斺 self.current_chain = cfg.get("default_chain", "esp32") self._lock = threading.Lock() @@ -341,9 +631,29 @@ def _build_cmd(self, name): if name == "demo": subst = {"{prompt}": self.current_prompt, "{device}": self.current_device} cmd = [subst.get(x, x) for x in cmd] - elif name == "rokid": + elif name == "harness": + # 璇佷功鐢ㄧ粷瀵硅矾寰勶細harness 鐨 cwd 鏄 OpenGlass 浠撳簱鏍癸紝 + # 鑰岃瘉涔︾敱 _ensure_certs 鐢熸垚鍦 minicpm_demo_dir/certs 涓嬶紙gateway 鍏辩敤锛夈 + _base = self.cfg.get("minicpm_demo_dir") or "." + subst = { + "{certfile}": os.path.join(_base, "certs", "cert.pem"), + "{keyfile}": os.path.join(_base, "certs", "key.pem"), + } + cmd = [subst.get(x, x) for x in cmd] + elif name == "demo_funnel": + # esp32_runtime锛歱rompt + 璁惧鍙傛暟(IP/绔彛/鏃嬭浆) + 婕忔枟妗d綅 subst = {"{prompt}": self.current_prompt} cmd = [subst.get(x, x) for x in cmd] + cmd += self._device_args() + cmd += self._funnel_extra_args() + elif name == "rokid": + _loc = self.cfg.get("local", {}) + subst = { + "{prompt}": self.current_prompt, + "{glasses_ssid}": _loc.get("glasses_ssid", ""), + "{glasses_psk}": _loc.get("glasses_psk", ""), + } + cmd = [subst.get(x, x) for x in cmd] cmd += self._rokid_extra_args() # 鍐呰仈 ps1 鐨 --enable-funasr 鍒嗘敮 return self._wrap_conda(cmd) @@ -436,6 +746,110 @@ def _rokid_env(self): self._log("rokid", f"鏃ュ織: {env['ROKID_V7_LOG_FILE']}") return env + def _device_args(self): + """浠 devices.json 鍙栧綋鍓嶇溂闀滅殑 IP / 绔彛 / 鏃嬭浆瑙掞紝濉粰 esp32_runtime銆""" + d = (self.cfg.get("device_map") or {}).get(self.current_device) + if not d or not d.get("host"): + self._log("demo_funnel", + f"!! devices.json 閲屾壘涓嶅埌銆寋self.current_device}銆嶇殑 esp32_host") + return [] + args = ["--esp32-host", d["host"], "--esp32-port", str(d.get("port", 80))] + if d.get("rotate"): + args += ["--rotate", str(d["rotate"])] + return args + + def _funnel_extra_args(self): + """鎸夐摼璺殑妗d綅杩藉姞婕忔枟鍙傛暟銆 + + 0 璇煶鎺у埗 涓嶅姞 鈫 璧 no_funnel 鍒嗘敮锛屾瘡杞彇涓甯х洿鍙 + 2 璐ㄩ噺绛涢 --funnel --no-reject 鈫 閫 best 浣嗘案杩滄斁琛 + 3 瀹屾暣闃插够瑙 --funnel --force-measure 鈫 閫 best + 鎷掔粷 + 璇煶鎻愮ず + """ + lv = int(self.chain().get("funnel", 0)) + if lv <= 0: + return [] + scene = self.cfg["scenes"].get(self.current_scene, "medicine") + args = ["--funnel", "--scene", scene] + if lv == 2: + args += ["--no-reject"] + else: + args += ["--force-measure", "--reject-wav-dir", self._reject_wav_dir()] + return args + + def _check_deps(self): + """鈶♀憿鈶 鍚姩鍓嶆鏌ュ叧閿緷璧栵紝缂轰簡灏辫娓呮缂轰粈涔堛佹庝箞瑁呫 + + 涓轰粈涔堝煎緱鍗曠嫭鏌ワ細paddleocr 瑁呬笉涓婃椂锛屾柟鍚戝垎绫诲櫒浼**闈欓粯鍥為**鍒 + 鏃╂湡鐨勭畝鏄撳垽鎹紝琛ㄧ幇鏄敾闈㈡槑鏄庢槸姝g殑鍗翠竴鐩存姤"鐢婚潰濂藉儚鍙嶄簡"銆 + 鎺ョ潃鎾姤鎶婂抚闂撮殧鎷夐暱鍙堣鎶"鏅冨姩" 鈥斺 鍏ㄧ▼涓嶆姤閿欙紝鏋侀毦瀹氫綅銆 + 瀹炴祴韪╄繃锛歱anel 鍦ㄦ病瑁呭ソ paddle 鐨勭幆澧冮噷鍚姩锛屼竴涓婃潵灏卞叏鏄 orient_flipped銆 + """ + lv = int(self.chain().get("funnel", 0)) + need = [("aiohttp", "aiohttp"), ("numpy", "numpy"), + ("PIL", "Pillow"), ("sounddevice", "sounddevice")] + if self.chain().get("tail") == "demo_funnel": + need += [("fastapi", "fastapi"), ("uvicorn", "uvicorn"), + ("yaml", "PyYAML"), ("funasr", "funasr")] + if lv >= 2: + need += [("cv2", "opencv-python")] + if lv >= 3 or lv == 2: + need += [("paddle", "paddlepaddle"), ("paddleocr", "paddleocr")] + import importlib.util as _iu + missing = [pip for mod, pip in need if _iu.find_spec(mod) is None] + if missing: + tail = self.chain().get("tail") or "demo" + self._log(tail, "!! 缂哄皯渚濊禆: " + ", ".join(missing)) + self._log(tail, " 璇峰湪**鍚姩 panel 鐨勯偅涓 conda 鐜**閲屽畨瑁咃細") + self._log(tail, " pip install " + " ".join(missing)) + self._log(tail, " 锛堝畬鏁存竻鍗曡 extensions/requirements-phase-b.txt锛") + return False + return True + + def _check_extensions(self): + """鈶♀憿鈶 闇瑕 OpenGlass 浠撳簱鍐呯殑 extensions/ 鍖呭畬鏁淬 + + 涓嶉渶瑕佸鍒跺埌鍒 鈥斺 _spawn 鎶 cwd 璁炬垚 OpenGlass 浠撳簱鏍广 + 缂烘枃浠舵椂瀛愯繘绋嬩細璧锋潵绔嬪埢姝汇佹棩蹇楅噷涓琛 No module named extensions锛 + 闈㈡澘涓婂彧鐪嬪埌鐏彉绾紝鎵浠ユ彁鍓嶆嫤浣忓苟璇存竻妤氥 + """ + if self.chain().get("tail") != "demo_funnel": + return True + f = os.path.join(self._repo_root(), "extensions", "assistive_harness", + "phase_b", "esp32_runtime.py") + if not os.path.isfile(f): + self._log("demo_funnel", + f"!! 鏈壘鍒 {f}\n" + f" 鈶♀憿鈶 闇瑕佷粨搴撳唴鐨 extensions/ 鍖咃紝璇风‘璁ゅ畠娌℃湁琚垹闄ゆ垨绉诲姩銆") + return False + return True + + def _repo_root(self): + """OpenGlass 浠撳簱鏍癸紙panel.py 鍦 runtime/openglass_omni/ 涓嬶紝寰涓婁笁绾э級銆""" + return os.path.dirname(os.path.dirname( + os.path.dirname(os.path.abspath(__file__)))) + + def _reject_wav_dir(self): + d = self.cfg.get("reject_wav_dir", + "extensions/assistive_harness/phase_b/assets/reject_wav") + return d if os.path.isabs(d) else os.path.join(self._repo_root(), d) + + def _check_reject_wav(self): + """妗d綅鈶h鎾彁绀洪煶锛寃av 蹇呴』浜嬪厛鐢 gen_reject_wavs.py 鐢熸垚濂 + 锛堜笖瑕佸湪 duplex 娌¤窇鐨勬椂鍊欑敓鎴愶級銆傝繖閲屽彧妫鏌ワ紝涓嶅湪闈㈡澘閲岀幇鍦虹敓鎴愩""" + if int(self.chain().get("funnel", 0)) < 3: + return True + full = self._reject_wav_dir() + try: + n = len([x for x in os.listdir(full) if x.lower().endswith(".wav")]) + except Exception: + n = 0 + if n == 0: + self._log("demo_funnel", + f"!! {full} 閲屾病鏈 wav銆傛。浣嶁懀瑕佹挱鎻愮ず闊筹紝" + f"璇峰厛鍦 duplex 鏈繍琛屾椂璺 gen_reject_wavs.py 鐢熸垚銆") + return False + return True + def _rokid_extra_args(self): """鎸夐厤缃拷鍔犲弬鏁帮紙瀵瑰簲 ps1 閲岀殑 $bridgeArgs += ...锛夈""" extra = [] @@ -524,7 +938,18 @@ def _spawn(self, name): creationflags = subprocess.CREATE_NEW_PROCESS_GROUP # worker / gateway 鏄笂娓 MiniCPM-o-Demo 鐨勬枃浠讹紙worker.py / gateway.py锛夛紝 # 蹇呴』鍦ㄩ偅涓洰褰曚笅鍚姩鎵嶈兘鎵惧埌銆俵lama / demo / rokid 鐢ㄧ殑鏄粷瀵硅矾寰勶紝涓嶄緷璧 cwd銆 - if name in ("worker", "gateway"): + # harness / demo_funnel 鐢 `python -m extensions...` 鍚姩锛 + # **cwd 蹇呴』鏄 OpenGlass 浠撳簱鏍** 鈥斺 extensions/ 灏卞湪瀹冧笅闈€ + # 鈽 涓嶈鐢 PYTHONPATH 浠f浛锛歚-m` 鏃 sys.path[0] 鏄 cwd锛宑wd 浼樺厛浜 + # PYTHONPATH銆傚鏋 cwd 璁炬垚 MiniCPM-o-Demo 鑰岄偅杈**涔熸湁**涓浠 + # extensions/锛孫penGlass 杩欎唤浼氳瀹屽叏閬斀锛堟敼浜嗕笉鐢熸晥锛夛紱 + # 鏇寸碂鐨勬槸 PYTHONPATH 鎶 OpenGlass 鏍瑰杩 sys.path锛屽疄娴嬪鑷 + # paddle 瀵煎叆澶辫触锛坧artially initialized module 'paddle' ... + # circular import锛夆啋 鏂瑰悜鍒嗙被鍣ㄥ姞杞藉け璐 鈫 鍥為鍒伴敊璇殑鏃╂湡 CV 鍒ゆ嵁 + # 鈫 绗竴杞氨璇姤 orient_flipped锛岀劧鍚庢挱鎶ユ媺闀垮抚闂撮殧鍙堣鎶 severe_shake銆 + if name in ("harness", "demo_funnel"): + _cwd = self._repo_root() + elif name in ("worker", "gateway"): _cwd = self.cfg.get("minicpm_demo_dir") or self.cfg.get("cwd") or None if not _cwd: self._log(name, "!! 鏈厤缃 minicpm_demo_dir锛圡iniCPM-o-Demo 鐩綍锛夛紝" @@ -559,6 +984,7 @@ def _spawn(self, name): return proc def _kill(self, name, graceful=False, grace_timeout=20): + self._forget_tail_fp(name) # 杩涚▼瑕佹病浜嗭紝鎸囩汗涓骞朵綔搴 proc = self.procs.get(name) if not proc or proc.poll() is not None: self.status[name] = "stopped" @@ -676,6 +1102,21 @@ def _wait_ready_or_die(self, name, timeout): self._log(name, "灏辩华(鍏抽敭瀛楀懡涓)锛屾斁琛") return True elapsed = time.time() - t_start + if name == "harness": + # 8021 鏄嚜绛 wss锛屼笉鑳界敤 HTTP health锛屽彧鑳芥帰 TCP 绔彛銆 + # 鑰屼笖绔彛寮浜嗕箣鍚 /ws/control 杩樿涓灏忎細鍎挎墠缁戝ソ锛 + # 涓嶇瓑鐨勮瘽 esp32_runtime 杩炶繃鍘讳細澶辫触銆 + if self._port_open(8021): + self._log("harness", "8021 绔彛宸插紑锛屽啀绛 2s 璁 /ws/control 缁戝ソ") + time.sleep(2.0) + self._log("harness", "harness 灏辩华锛屾斁琛") + return True + _now = time.time() + if _now - last_log > 5: + last_log = _now + self._log("harness", f"绛夊緟 8021 (FunASR 妯″瀷鍔犺浇涓) {elapsed:.0f}s") + time.sleep(1.0) + continue if name == "llama": # llama-omni-server锛氳疆璇 /health 杩斿洖 200 鎵嶆斁琛岋紙= 浣犳墜鍔ㄧ殑 curl .../health锛 if self._http_ok(llama_url): @@ -747,6 +1188,9 @@ def _start_one(self, name): self._spawn(name) if self._wait_ready_or_die(name, ready_timeout): self.status[name] = "running" + if name == self.tail(): + # 璁颁笅杩欐鏄寜鍝閰嶇疆璧风殑锛屼笅娆″垏閾捐矾鏃剁敤鏉ュ垽鏂涓嶈閲嶅惎 + self._tail_started_for[name] = self._tail_fingerprint() return True # 鏈氨缁/閫鍑/琚ュ仠锛氬厛鎶婅繖娆 spawn 鐨勮繘绋嬫潃骞插噣锛岀粷涓嶇暀娈嬩綑 p = self.procs.get(name) @@ -769,12 +1213,34 @@ def _start_one(self, name): return False return False + def _forget_tail_fp(self, name): + """杩涚▼涓嶅湪浜嗗氨娓呮帀鎸囩汗锛屽惁鍒欎笅娆′細璇垽鎴"閰嶇疆娌″彉銆佽烦杩囧惎鍔"銆""" + self._tail_started_for.pop(name, None) + def _other_tails(self): """闄ゅ綋鍓嶉摼璺锛屽叾瀹冮摼璺殑绗洓绾ц繘绋嬪悕銆""" cur = self.tail() return [c["tail"] for k, c in self.cfg["chains"].items() if c["tail"] != cur] + def _tail_fingerprint(self): + """褰撳墠閾捐矾涓嬶紝灏捐繘绋嬬湡姝d緷璧栫殑閭e嚑涓噺銆 + + 浠讳綍涓椤瑰彉浜嗭紝宸插湪璺戠殑灏捐繘绋嬪氨鏄寜鏃ч厤缃捣鐨勶紝蹇呴』閲嶅惎锛 + chain 鍐冲畾婕忔枟妗d綅锛--funnel / --no-reject / --force-measure锛 + scene 鍐冲畾 --scene锛堜弗鏍/鏃ュ父鍒ゆ嵁锛 + device 鍐冲畾 --esp32-host / --esp32-port / --rotate + prompt 鍐冲畾 --prompt + """ + c = self.chain() + return (self.current_chain, c.get("funnel", 0), self.current_scene, + self.current_device, self.current_prompt) + def _do_start_all(self): + # 鈶♀憿鈶 鐨勪袱涓墠缃鏌ワ細extensions 澶嶅埗浜嗘病銆佹。浣嶁懀鐨 wav 鏈夋病鏈夈 + # 鎻愬墠鎷︿綇姣旇窇璧锋潵鎵嶅彂鐜板ソ 鈥斺 鍚庤呮椂 duplex 宸插湪璺戯紝娌℃硶褰撳満琛ャ + if (not self._check_extensions() or not self._check_deps() + or not self._check_reject_wav()): + return """鎸夊簭琛ラ綈锛氬彧璧锋病鍦ㄨ窇鐨勶紝宸插湪璺戠殑璺宠繃锛涜鎬ュ仠绔嬪嵆鍋滀笅銆""" # 鍒囬摼鍚庤嫢鍙︿竴鏉¢摼鐨勫熬宸磋繕娲荤潃锛屽厛鏉鎺夆斺斾袱涓鎴风鍚屾椂鍗犱竴涓 gateway # 浼氫簰鐩告姠 session锛屽繀椤讳簰鏂ャ @@ -787,9 +1253,22 @@ def _do_start_all(self): if self._cancel.is_set(): return if self._alive(name): - self.status[name] = "running" - self._log(name, "宸插湪杩愯锛岃烦杩囧惎鍔") - continue + # 灏捐繘绋嬭繕瑕佹瘮閰嶇疆鎸囩汗锛氬悓涓涓 demo_funnel 杩涚▼锛 + # 鈶♀憿鈶 浼犵殑鍙傛暟瀹屽叏涓嶅悓锛屽厜鐪"娲荤潃"浼氭紡鎺夐噸鍚 + if name == self.tail(): + want = self._tail_fingerprint() + got = self._tail_started_for.get(name) + if got is not None and got != want: + self._log(name, "閰嶇疆宸插彉锛堥摼璺/鍒ゆ嵁/鐪奸暅/Prompt锛夛紝閲嶅惎璇ヨ繘绋") + self._kill(name, graceful=True, grace_timeout=120) + else: + self.status[name] = "running" + self._log(name, "宸插湪杩愯锛岃烦杩囧惎鍔") + continue + else: + self.status[name] = "running" + self._log(name, "宸插湪杩愯锛岃烦杩囧惎鍔") + continue if not self._start_one(name): if self._cancel.is_set(): self._log(name, "!! 鍚姩琚ュ仠涓") @@ -940,24 +1419,34 @@ def get_config(self): "need_device": c["need_device"], "fpv_url": self.cfg.get(c["fpv_key"], ""), "procs": c["start_order"], + # 鍙湁寮浜嗘紡鏂楃殑閾捐矾(鈶⑩懀)鎵嶉渶瑕侀夊垽鎹。浣嶏紱鈶犫憽閫変簡涔熶笉璧蜂綔鐢 + "need_scene": int(c.get("funnel", 0)) > 0, } return { "presets": self.cfg["presets"], "fpv_url": self.cfg["fpv_url"], "devices": self.cfg.get("devices", []), "chains": chains, + "scenes": list(self.cfg.get("scenes", {}).keys()), + "default_scene": self.cfg.get("default_scene", ""), + "proc_labels": self.cfg.get("proc_labels", {}), "default_chain": self.cfg.get("default_chain", "esp32"), } def set_chain(self, chain): return self.mgr.set_chain(chain) + def set_scene(self, scene): + if scene: + self.mgr.current_scene = scene + return "ok" + def set_device(self, device): if device: self.mgr.current_device = device return "ok" - def start_all(self, prompt=None, device=None, chain=None): + def start_all(self, prompt=None, device=None, chain=None, scene=None): # prompt 鏉ヨ嚜鍓嶇鏂囨湰妗嗭紙鐢ㄦ埛閫夌殑 preset 鎴栫紪杈戝悗鐨勫唴瀹癸級銆傚湪璧疯繘绋嬪墠鏇存柊 # current_prompt锛屽惁鍒欎竴閿惎鍔ㄥ彧浼氱敤鍒濆鐨勭涓涓 preset銆 if prompt: @@ -966,6 +1455,8 @@ def start_all(self, prompt=None, device=None, chain=None): self.mgr.set_chain(chain) if device: self.mgr.current_device = device + if scene: + self.mgr.current_scene = scene self.mgr.start_all(); return "ok" def stop_all(self): diff --git a/runtime/openglass_omni/runtime.example.json b/runtime/openglass_omni/runtime.example.json index f3c77a1..c69bde1 100644 --- a/runtime/openglass_omni/runtime.example.json +++ b/runtime/openglass_omni/runtime.example.json @@ -1,33 +1,28 @@ { - "_comment": "Copy to runtime.local.json. Paths are resolved relative to this file and may use environment variables.", - "python": "", - "minicpm_demo_root": "${MINICPM_O_DEMO_ROOT}", - "llama_cpp_omni_root": "${LLAMA_CPP_OMNI_ROOT}", - "devices_file": "devices.local.json", - "prompts_file": "prompts.json", - "state_dir": "state", - "worker": { - "host": "127.0.0.1", - "port": 22440, - "ready_timeout_s": 180 + "_comment": "澶嶅埗涓 runtime.local.json 鍚庡~鍐欍傝鏂囦欢涓嶈繘浠撳簱锛堝姞杩 .gitignore锛夈傝矾寰勬敮鎸 ${ENV_VAR} 鍜 ~銆", + + "_conda": "鐣欑┖琛ㄧず宸插湪澶栭儴 conda activate锛涘~鐜鍚嶅垯鐢 panel 鐢 conda run 鍚姩瀛愯繘绋", + "conda_env": null, + + "_paths": "涓変釜浠撳簱鐩镐簰鐙珛锛屽悇鑷~鑷繁鐨勭粷瀵硅矾寰", + "minicpm_demo_root": "D:\\path\\to\\MiniCPM-o-Demo", + "llama_server": "D:\\path\\to\\llama.cpp-omni\\build\\bin\\Release\\llama-omni-server.exe", + "llama_model": "D:\\path\\to\\MiniCPM-o-gguf\\MiniCPM-o-4_5-Q4_K_M.gguf", + + "_asr": "鈶♀憿鈶 閾捐矾鐨勮闊虫帶鍒(8021 harness)鐢ㄧ殑 FunASR 娴佸紡妯″瀷鐩綍", + "asr_model": "D:\\path\\to\\LocalASRmodel\\speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-online", + + "_glasses": "Rokid 閾捐矾瑕佹妸 WiFi 浼犵粰 APK銆備笉瑕佹妸鐪熷疄瀵嗙爜鎻愪氦杩涗粨搴", + "glasses": { + "ssid": "YOUR_WIFI_SSID", + "psk": "YOUR_WIFI_PASSWORD" }, - "gateway": { - "host": "127.0.0.1", - "bind_host": "0.0.0.0", - "port": 8040, - "tls": true, - "certfile": "certs/cert.pem", - "keyfile": "certs/key.pem", - "ready_timeout_s": 45 - }, - "demo": { - "ui_host": "127.0.0.1", - "ui_port": 8080, - "audio_endpoint": "/ws_audio", - "image_min_interval_s": 0.8, - "player_hostapi": "wasapi", - "player_prebuffer_ms": 200, - "connect_wait_s": 120, - "ready_timeout_s": 20 - } + + "_ports": "浠ヤ笅涓哄綋鍓 V2 閾捐矾鐨勫疄闄呯鍙o紝浠呬緵鍙傝冿紱panel.py 閲屽凡鎸夋閰嶇疆锛屾敼鍔ㄩ渶鍚屾鏀 panel", + "worker": { "host": "127.0.0.1", "port": 22400 }, + "gateway": { "host": "127.0.0.1", "port": 8006, "tls": true, + "certfile": "certs/cert.pem", "keyfile": "certs/key.pem" }, + "llama": { "host": "127.0.0.1", "port": 22500 }, + "harness": { "host": "127.0.0.1", "port": 8021 }, + "demo": { "ui_port": 8080 } }