149 lines
3.4 KiB
JavaScript
149 lines
3.4 KiB
JavaScript
|
|
const speech = require('@google-cloud/speech');
|
|
const portAudio = require('naudiodon');
|
|
// const rs = fs.createReadStream('rawAudio.wav');
|
|
|
|
// Creates a client
|
|
const client = new speech.SpeechClient();
|
|
|
|
const encoding = 'LINEAR16';
|
|
const sampleRateHertz = 16000;
|
|
const languageCode = 'en-US';
|
|
|
|
const config = {
|
|
encoding: encoding,
|
|
sampleRateHertz: sampleRateHertz,
|
|
languageCode: languageCode,
|
|
};
|
|
|
|
/**
|
|
* Note that transcription is limited to 60 seconds audio.
|
|
* Use a GCS file for audio longer than 1 minute.
|
|
*/
|
|
|
|
async function transcribeSpeech (audio) {
|
|
const request = {
|
|
config: config,
|
|
audio: audio,
|
|
};
|
|
|
|
// Detects speech in the audio file. This creates a recognition job that you
|
|
// can wait for now, or get its result later.
|
|
const [operation] = await client.longRunningRecognize(request);
|
|
|
|
// Get a Promise representation of the final result of the job
|
|
const [response] = await operation.promise();
|
|
|
|
const transcription = response.results
|
|
.map(result => result.alternatives[0].transcript)
|
|
.join('\n');
|
|
console.log(`Transcription: ${transcription}`);
|
|
}
|
|
|
|
const audioInput = [];
|
|
// Create an instance of AudioIO with inOptions (defaults are as below), which will return a ReadableStream
|
|
const ia = new portAudio.AudioIO({
|
|
inOptions: {
|
|
channelCount: 1,
|
|
sampleFormat: 16,
|
|
sampleRate: 16000,
|
|
deviceId: -1,
|
|
closeOnError: false,
|
|
},
|
|
});
|
|
ia.setEncoding('base64');
|
|
ia.start();
|
|
ia.on('data', (chunk) => {
|
|
const buf = Buffer.from(chunk, 'base64');
|
|
audioInput.push(buf);
|
|
});
|
|
|
|
const ao = new portAudio.AudioIO({
|
|
outOptions: {
|
|
sampleFormat: 16,
|
|
channelCount: 1,
|
|
sampleRate: 16000,
|
|
deviceId: -1,
|
|
closeOnError: false,
|
|
}
|
|
});
|
|
ao.start();
|
|
|
|
async function processAudio() {
|
|
|
|
const buf = Buffer.concat(audioInput);
|
|
play(audioInput);
|
|
|
|
while(audioInput.length) { audioInput.pop() }
|
|
|
|
transcribeSpeech({
|
|
content: buf
|
|
});
|
|
}
|
|
|
|
|
|
function play(input) {
|
|
let i = 0;
|
|
// format buffers into half their size to account for writable highwaterMark.
|
|
const audio = bufSplit(input);
|
|
// call this fuction after last audio chunk has been written.
|
|
const callback = () => {
|
|
// TODO: clear portAudio writable buffer on write end.
|
|
}
|
|
write();
|
|
// iterate through audio array and write buffers to portAudio writable.
|
|
function write() {
|
|
let chunk;
|
|
let ok = true;
|
|
do {
|
|
chunk = audio[i];
|
|
if (i === audio.length - 1) {
|
|
// write last chunk.
|
|
ao.write(chunk, null, callback);
|
|
} else {
|
|
// check for backpreassure.
|
|
ok = ao.write(chunk, null);
|
|
}
|
|
i++;
|
|
} while (i < audio.length && ok);
|
|
|
|
if (i < audio.length) {
|
|
// Had to stop early!
|
|
// Write some more once it drains.
|
|
ao.once('drain', write);
|
|
}
|
|
}
|
|
}
|
|
|
|
// utility function used by play func
|
|
function bufSplit(input){
|
|
const result = [];
|
|
input.forEach((b) => {
|
|
// split buffer into two.
|
|
result.push(b.slice(0, b.length / 2), b.slice(b.length / 2, b.length));
|
|
});
|
|
return result;
|
|
}
|
|
|
|
setTimeout(async () => {
|
|
ia.pause();
|
|
processAudio();
|
|
}, 3000);
|
|
|
|
setTimeout(() => {
|
|
ia.resume();
|
|
}, 5000)
|
|
|
|
setTimeout(async () => {
|
|
ia.pause();
|
|
processAudio();
|
|
}, 8000);
|
|
|
|
setTimeout(() => {
|
|
ia.resume();
|
|
}, 10000)
|
|
|
|
setTimeout(async () => {
|
|
ia.pause();
|
|
processAudio();
|
|
}, 15000);
|