remove vad from plugin
This commit is contained in:
@@ -4,9 +4,7 @@
|
|||||||
"name": "Audio Transcription",
|
"name": "Audio Transcription",
|
||||||
"version": "1.0.0",
|
"version": "1.0.0",
|
||||||
"description": "This extension captures the audio on the current tab, sends it to a server for transcription and shows the transcription in Real-time.",
|
"description": "This extension captures the audio on the current tab, sends it to a server for transcription and shows the transcription in Real-time.",
|
||||||
"content_security_policy": {
|
|
||||||
"extension_pages": "script-src 'self' 'wasm-unsafe-eval'; object-src 'self';"
|
|
||||||
},
|
|
||||||
"options_page": "options.html",
|
"options_page": "options.html",
|
||||||
"background": {
|
"background": {
|
||||||
"service_worker": "background.js"
|
"service_worker": "background.js"
|
||||||
|
|||||||
@@ -73,43 +73,13 @@ function resampleTo16kHZ(audioData, origSampleRate = 44100) {
|
|||||||
*/
|
*/
|
||||||
async function startRecord(option) {
|
async function startRecord(option) {
|
||||||
const stream = await captureTabAudio();
|
const stream = await captureTabAudio();
|
||||||
var doVad = true;
|
|
||||||
if (stream) {
|
if (stream) {
|
||||||
// call when the stream inactive
|
// call when the stream inactive
|
||||||
stream.oninactive = () => {
|
stream.oninactive = () => {
|
||||||
window.close();
|
window.close();
|
||||||
};
|
};
|
||||||
|
|
||||||
// create onnx model
|
|
||||||
// initialize onnx model
|
|
||||||
const session = await ort.InferenceSession.create('./silero_vad.onnx');
|
|
||||||
var h = new Array(128);
|
|
||||||
for (let i = 0; i < h.length; i++) {
|
|
||||||
h[i] = 0;
|
|
||||||
}
|
|
||||||
var c = new Array(128);
|
|
||||||
for (let i = 0; i < h.length; i++) {
|
|
||||||
c[i] = 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
const sr = new BigInt64Array(1)
|
|
||||||
sr[0] = BigInt(16000);
|
|
||||||
const srate = new ort.Tensor('int64', sr, [1]);
|
|
||||||
let speech_prob = undefined;
|
|
||||||
const vad_infer = async (feed_dict) => {
|
|
||||||
// feed inputs and run
|
|
||||||
try{
|
|
||||||
const results = await session.run(feed_dict);
|
|
||||||
|
|
||||||
// update states
|
|
||||||
h = results.hn.data
|
|
||||||
c = results.cn.data
|
|
||||||
speech_prob = results.output.data
|
|
||||||
} catch(e) {
|
|
||||||
console.log(e)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const socket = new WebSocket("ws://localhost:9090/");
|
const socket = new WebSocket("ws://localhost:9090/");
|
||||||
socket.onopen = function(e) {
|
socket.onopen = function(e) {
|
||||||
socket.send("handshake");
|
socket.send("handshake");
|
||||||
@@ -136,21 +106,8 @@ async function startRecord(option) {
|
|||||||
|
|
||||||
audioDataCache.push(inputData);
|
audioDataCache.push(inputData);
|
||||||
|
|
||||||
// voice activity detection inference
|
|
||||||
const audioBuffer = new ort.Tensor('float32', audioData16kHz, [1, audioData16kHz.length]);
|
|
||||||
const hh = new ort.Tensor('float32', h, [2, 1, 64]);
|
|
||||||
const hc = new ort.Tensor('float32', c, [2, 1, 64]);
|
|
||||||
const feeds = { input: audioBuffer, sr: srate, h: hh, c: hc};
|
|
||||||
|
|
||||||
// feed inputs and run
|
// feed inputs and run
|
||||||
if (doVad) {
|
socket.send(audioData16kHz);
|
||||||
vad_infer(feeds)
|
|
||||||
if (speech_prob > 0.4) {
|
|
||||||
socket.send(audioData16kHz);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
console.log("no speech found: " + speech_prob)
|
|
||||||
}
|
|
||||||
};
|
};
|
||||||
|
|
||||||
// Prevent page mute
|
// Prevent page mute
|
||||||
|
|||||||
Binary file not shown.
Vendored
-6
File diff suppressed because one or more lines are too long
Binary file not shown.
Reference in New Issue
Block a user