mirror of
https://repo.dactyloidae.xyz/Dactyloidae/UXP.git
synced 2026-10-06 23:37:31 +09:00
import FIREFOX_52_6_0esr_RELEASE from mozilla-esr52 hg repo
This commit is contained in:
commit
dcd9973243
150858 changed files with 23884658 additions and 0 deletions
9
dom/media/webspeech/moz.build
Normal file
9
dom/media/webspeech/moz.build
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
# vim: set filetype=python:
|
||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
DIRS = ['synth']
|
||||
|
||||
if CONFIG['MOZ_WEBSPEECH']:
|
||||
DIRS += ['recognition']
|
||||
|
|
@ -0,0 +1,357 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsThreadUtils.h"
|
||||
#include "nsXPCOMCIDInternal.h"
|
||||
#include "PocketSphinxSpeechRecognitionService.h"
|
||||
#include "nsIFile.h"
|
||||
#include "SpeechGrammar.h"
|
||||
#include "SpeechRecognition.h"
|
||||
#include "SpeechRecognitionAlternative.h"
|
||||
#include "SpeechRecognitionResult.h"
|
||||
#include "SpeechRecognitionResultList.h"
|
||||
#include "nsIObserverService.h"
|
||||
#include "MediaPrefs.h"
|
||||
#include "mozilla/Services.h"
|
||||
#include "nsDirectoryServiceDefs.h"
|
||||
#include "nsDirectoryServiceUtils.h"
|
||||
#include "nsMemory.h"
|
||||
|
||||
extern "C" {
|
||||
#include "pocketsphinx/pocketsphinx.h"
|
||||
#include "sphinxbase/logmath.h"
|
||||
#include "sphinxbase/sphinx_config.h"
|
||||
#include "sphinxbase/jsgf.h"
|
||||
}
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
using namespace dom;
|
||||
|
||||
class DecodeResultTask : public Runnable
|
||||
{
|
||||
public:
|
||||
DecodeResultTask(const nsString& hypstring,
|
||||
float64 confidence,
|
||||
WeakPtr<dom::SpeechRecognition> recognition)
|
||||
: mResult(hypstring),
|
||||
mConfidence(confidence),
|
||||
mRecognition(recognition),
|
||||
mWorkerThread(do_GetCurrentThread())
|
||||
{
|
||||
MOZ_ASSERT(
|
||||
!NS_IsMainThread()); // This should be running on the worker thread
|
||||
}
|
||||
|
||||
NS_IMETHOD
|
||||
Run() override
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread()); // This method is supposed to run on the main
|
||||
// thread!
|
||||
|
||||
// Declare javascript result events
|
||||
RefPtr<SpeechEvent> event = new SpeechEvent(
|
||||
mRecognition, SpeechRecognition::EVENT_RECOGNITIONSERVICE_FINAL_RESULT);
|
||||
SpeechRecognitionResultList* resultList =
|
||||
new SpeechRecognitionResultList(mRecognition);
|
||||
SpeechRecognitionResult* result = new SpeechRecognitionResult(mRecognition);
|
||||
if (0 < mRecognition->MaxAlternatives()) {
|
||||
SpeechRecognitionAlternative* alternative =
|
||||
new SpeechRecognitionAlternative(mRecognition);
|
||||
|
||||
alternative->mTranscript = mResult;
|
||||
alternative->mConfidence = mConfidence;
|
||||
|
||||
result->mItems.AppendElement(alternative);
|
||||
}
|
||||
resultList->mItems.AppendElement(result);
|
||||
|
||||
event->mRecognitionResultList = resultList;
|
||||
NS_DispatchToMainThread(event);
|
||||
|
||||
// If we don't destroy the thread when we're done with it, it will hang
|
||||
// around forever... bad!
|
||||
// But thread->Shutdown must be called from the main thread, not from the
|
||||
// thread itself.
|
||||
return mWorkerThread->Shutdown();
|
||||
}
|
||||
|
||||
private:
|
||||
nsString mResult;
|
||||
float64 mConfidence;
|
||||
WeakPtr<dom::SpeechRecognition> mRecognition;
|
||||
nsCOMPtr<nsIThread> mWorkerThread;
|
||||
};
|
||||
|
||||
class DecodeTask : public Runnable
|
||||
{
|
||||
public:
|
||||
DecodeTask(WeakPtr<dom::SpeechRecognition> recogntion,
|
||||
const nsTArray<int16_t>& audiovector, ps_decoder_t* ps)
|
||||
: mRecognition(recogntion), mAudiovector(audiovector), mPs(ps)
|
||||
{
|
||||
}
|
||||
|
||||
NS_IMETHOD
|
||||
Run() override
|
||||
{
|
||||
char const* hyp;
|
||||
int rv;
|
||||
int32 final;
|
||||
int32 logprob;
|
||||
float64 confidence;
|
||||
nsAutoCString hypoValue;
|
||||
|
||||
rv = ps_start_utt(mPs);
|
||||
rv = ps_process_raw(mPs, &mAudiovector[0], mAudiovector.Length(), FALSE,
|
||||
FALSE);
|
||||
|
||||
rv = ps_end_utt(mPs);
|
||||
confidence = 0;
|
||||
if (rv >= 0) {
|
||||
hyp = ps_get_hyp_final(mPs, &final);
|
||||
if (hyp && final) {
|
||||
logprob = ps_get_prob(mPs);
|
||||
confidence = logmath_exp(ps_get_logmath(mPs), logprob);
|
||||
hypoValue.Assign(hyp);
|
||||
}
|
||||
}
|
||||
|
||||
nsCOMPtr<nsIRunnable> resultrunnable =
|
||||
new DecodeResultTask(NS_ConvertUTF8toUTF16(hypoValue), confidence, mRecognition);
|
||||
return NS_DispatchToMainThread(resultrunnable);
|
||||
}
|
||||
|
||||
private:
|
||||
WeakPtr<dom::SpeechRecognition> mRecognition;
|
||||
nsTArray<int16_t> mAudiovector;
|
||||
ps_decoder_t* mPs;
|
||||
};
|
||||
|
||||
NS_IMPL_ISUPPORTS(PocketSphinxSpeechRecognitionService,
|
||||
nsISpeechRecognitionService, nsIObserver)
|
||||
|
||||
PocketSphinxSpeechRecognitionService::PocketSphinxSpeechRecognitionService()
|
||||
{
|
||||
mSpeexState = nullptr;
|
||||
|
||||
// get root folder
|
||||
nsCOMPtr<nsIFile> tmpFile;
|
||||
nsAutoString aStringAMPath; // am folder
|
||||
nsAutoString aStringDictPath; // dict folder
|
||||
|
||||
NS_GetSpecialDirectory(NS_GRE_DIR, getter_AddRefs(tmpFile));
|
||||
#if defined(XP_WIN) // for some reason, on windows NS_GRE_DIR is not bin root,
|
||||
// but bin/browser
|
||||
tmpFile->AppendRelativePath(NS_LITERAL_STRING(".."));
|
||||
#endif
|
||||
tmpFile->AppendRelativePath(NS_LITERAL_STRING("models"));
|
||||
tmpFile->AppendRelativePath(NS_LITERAL_STRING("en-US"));
|
||||
tmpFile->GetPath(aStringAMPath);
|
||||
|
||||
NS_GetSpecialDirectory(NS_GRE_DIR, getter_AddRefs(tmpFile));
|
||||
#if defined(XP_WIN) // for some reason, on windows NS_GRE_DIR is not bin root,
|
||||
// but bin/browser
|
||||
tmpFile->AppendRelativePath(NS_LITERAL_STRING(".."));
|
||||
#endif
|
||||
tmpFile->AppendRelativePath(NS_LITERAL_STRING("models")); //
|
||||
tmpFile->AppendRelativePath(NS_LITERAL_STRING("dict")); //
|
||||
tmpFile->AppendRelativePath(NS_LITERAL_STRING("en-US.dic")); //
|
||||
tmpFile->GetPath(aStringDictPath);
|
||||
|
||||
// FOR B2G PATHS HARDCODED (APPEND /DATA ON THE BEGINING, FOR DESKTOP, ONLY
|
||||
// MODELS/ RELATIVE TO ROOT
|
||||
mPSConfig = cmd_ln_init(nullptr, ps_args(), TRUE, "-bestpath", "yes", "-hmm",
|
||||
ToNewUTF8String(aStringAMPath), // acoustic model
|
||||
"-dict", ToNewUTF8String(aStringDictPath), nullptr);
|
||||
if (mPSConfig == nullptr) {
|
||||
ISDecoderCreated = false;
|
||||
} else {
|
||||
mPSHandle = ps_init(mPSConfig);
|
||||
if (mPSHandle == nullptr) {
|
||||
ISDecoderCreated = false;
|
||||
} else {
|
||||
ISDecoderCreated = true;
|
||||
}
|
||||
}
|
||||
|
||||
ISGrammarCompiled = false;
|
||||
}
|
||||
|
||||
PocketSphinxSpeechRecognitionService::~PocketSphinxSpeechRecognitionService()
|
||||
{
|
||||
if (mPSConfig) {
|
||||
free(mPSConfig);
|
||||
}
|
||||
if (mPSHandle) {
|
||||
free(mPSHandle);
|
||||
}
|
||||
|
||||
mSpeexState = nullptr;
|
||||
}
|
||||
|
||||
// CALL START IN JS FALLS HERE
|
||||
NS_IMETHODIMP
|
||||
PocketSphinxSpeechRecognitionService::Initialize(
|
||||
WeakPtr<SpeechRecognition> aSpeechRecognition)
|
||||
{
|
||||
if (!ISDecoderCreated || !ISGrammarCompiled) {
|
||||
return NS_ERROR_NOT_INITIALIZED;
|
||||
} else {
|
||||
mAudioVector.Clear();
|
||||
|
||||
if (mSpeexState) {
|
||||
mSpeexState = nullptr;
|
||||
}
|
||||
|
||||
mRecognition = aSpeechRecognition;
|
||||
nsCOMPtr<nsIObserverService> obs = services::GetObserverService();
|
||||
obs->AddObserver(this, SPEECH_RECOGNITION_TEST_EVENT_REQUEST_TOPIC, false);
|
||||
obs->AddObserver(this, SPEECH_RECOGNITION_TEST_END_TOPIC, false);
|
||||
return NS_OK;
|
||||
}
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
PocketSphinxSpeechRecognitionService::ProcessAudioSegment(
|
||||
AudioSegment* aAudioSegment, int32_t aSampleRate)
|
||||
{
|
||||
if (!mSpeexState) {
|
||||
mSpeexState = speex_resampler_init(1, aSampleRate, 16000,
|
||||
SPEEX_RESAMPLER_QUALITY_MAX, nullptr);
|
||||
}
|
||||
aAudioSegment->ResampleChunks(mSpeexState, aSampleRate, 16000);
|
||||
|
||||
AudioSegment::ChunkIterator iterator(*aAudioSegment);
|
||||
|
||||
while (!iterator.IsEnded()) {
|
||||
mozilla::AudioChunk& chunk = *(iterator);
|
||||
MOZ_ASSERT(chunk.mBuffer);
|
||||
const int16_t* buf = static_cast<const int16_t*>(chunk.mChannelData[0]);
|
||||
|
||||
for (int i = 0; i < iterator->mDuration; i++) {
|
||||
mAudioVector.AppendElement((int16_t)buf[i]);
|
||||
}
|
||||
iterator.Next();
|
||||
}
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
PocketSphinxSpeechRecognitionService::SoundEnd()
|
||||
{
|
||||
speex_resampler_destroy(mSpeexState);
|
||||
mSpeexState = nullptr;
|
||||
|
||||
// To create a new thread, get the thread manager
|
||||
nsCOMPtr<nsIThreadManager> tm = do_GetService(NS_THREADMANAGER_CONTRACTID);
|
||||
nsCOMPtr<nsIThread> decodethread;
|
||||
nsresult rv = tm->NewThread(0, 0, getter_AddRefs(decodethread));
|
||||
if (NS_FAILED(rv)) {
|
||||
// In case of failure, call back immediately with an empty string which
|
||||
// indicates failure
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
nsCOMPtr<nsIRunnable> r =
|
||||
new DecodeTask(mRecognition, mAudioVector, mPSHandle);
|
||||
decodethread->Dispatch(r, nsIEventTarget::DISPATCH_NORMAL);
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
PocketSphinxSpeechRecognitionService::ValidateAndSetGrammarList(
|
||||
SpeechGrammar* aSpeechGrammar,
|
||||
nsISpeechGrammarCompilationCallback* aCallback)
|
||||
{
|
||||
if (!ISDecoderCreated) {
|
||||
ISGrammarCompiled = false;
|
||||
} else if (aSpeechGrammar) {
|
||||
nsAutoString grammar;
|
||||
ErrorResult rv;
|
||||
aSpeechGrammar->GetSrc(grammar, rv);
|
||||
|
||||
int result = ps_set_jsgf_string(mPSHandle, "name",
|
||||
NS_ConvertUTF16toUTF8(grammar).get());
|
||||
|
||||
if (result != 0) {
|
||||
ISGrammarCompiled = false;
|
||||
} else {
|
||||
ps_set_search(mPSHandle, "name");
|
||||
|
||||
ISGrammarCompiled = true;
|
||||
}
|
||||
} else {
|
||||
ISGrammarCompiled = false;
|
||||
}
|
||||
|
||||
return ISGrammarCompiled ? NS_OK : NS_ERROR_NOT_INITIALIZED;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
PocketSphinxSpeechRecognitionService::Abort()
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
PocketSphinxSpeechRecognitionService::Observe(nsISupports* aSubject,
|
||||
const char* aTopic,
|
||||
const char16_t* aData)
|
||||
{
|
||||
MOZ_ASSERT(MediaPrefs::WebSpeechFakeRecognitionService(),
|
||||
"Got request to fake recognition service event, "
|
||||
"but " TEST_PREFERENCE_FAKE_RECOGNITION_SERVICE " is not set");
|
||||
|
||||
if (!strcmp(aTopic, SPEECH_RECOGNITION_TEST_END_TOPIC)) {
|
||||
nsCOMPtr<nsIObserverService> obs = services::GetObserverService();
|
||||
obs->RemoveObserver(this, SPEECH_RECOGNITION_TEST_EVENT_REQUEST_TOPIC);
|
||||
obs->RemoveObserver(this, SPEECH_RECOGNITION_TEST_END_TOPIC);
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
const nsDependentString eventName = nsDependentString(aData);
|
||||
|
||||
if (eventName.EqualsLiteral("EVENT_RECOGNITIONSERVICE_ERROR")) {
|
||||
mRecognition->DispatchError(
|
||||
SpeechRecognition::EVENT_RECOGNITIONSERVICE_ERROR,
|
||||
SpeechRecognitionErrorCode::Network, // TODO different codes?
|
||||
NS_LITERAL_STRING("RECOGNITIONSERVICE_ERROR test event"));
|
||||
|
||||
} else if (eventName.EqualsLiteral("EVENT_RECOGNITIONSERVICE_FINAL_RESULT")) {
|
||||
RefPtr<SpeechEvent> event = new SpeechEvent(
|
||||
mRecognition, SpeechRecognition::EVENT_RECOGNITIONSERVICE_FINAL_RESULT);
|
||||
|
||||
event->mRecognitionResultList = BuildMockResultList();
|
||||
NS_DispatchToMainThread(event);
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
SpeechRecognitionResultList*
|
||||
PocketSphinxSpeechRecognitionService::BuildMockResultList()
|
||||
{
|
||||
SpeechRecognitionResultList* resultList =
|
||||
new SpeechRecognitionResultList(mRecognition);
|
||||
SpeechRecognitionResult* result = new SpeechRecognitionResult(mRecognition);
|
||||
if (0 < mRecognition->MaxAlternatives()) {
|
||||
SpeechRecognitionAlternative* alternative =
|
||||
new SpeechRecognitionAlternative(mRecognition);
|
||||
|
||||
alternative->mTranscript = NS_LITERAL_STRING("Mock final result");
|
||||
alternative->mConfidence = 0.0f;
|
||||
|
||||
result->mItems.AppendElement(alternative);
|
||||
}
|
||||
resultList->mItems.AppendElement(result);
|
||||
|
||||
return resultList;
|
||||
}
|
||||
|
||||
} // namespace mozilla
|
||||
|
|
@ -0,0 +1,85 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_PocketSphinxRecognitionService_h
|
||||
#define mozilla_dom_PocketSphinxRecognitionService_h
|
||||
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsTArray.h"
|
||||
#include "nsIObserver.h"
|
||||
#include "nsISpeechRecognitionService.h"
|
||||
#include "speex/speex_resampler.h"
|
||||
|
||||
extern "C" {
|
||||
#include <pocketsphinx/pocketsphinx.h>
|
||||
#include <sphinxbase/sphinx_config.h>
|
||||
}
|
||||
|
||||
#define NS_POCKETSPHINX_SPEECH_RECOGNITION_SERVICE_CID \
|
||||
{ \
|
||||
0x0ff5ce56, 0x5b09, 0x4db8, { \
|
||||
0xad, 0xc6, 0x82, 0x66, 0xaf, 0x95, 0xf8, 0x64 \
|
||||
} \
|
||||
};
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
/**
|
||||
* Pocketsphix implementation of the nsISpeechRecognitionService interface
|
||||
*/
|
||||
class PocketSphinxSpeechRecognitionService : public nsISpeechRecognitionService,
|
||||
public nsIObserver
|
||||
{
|
||||
public:
|
||||
// Add XPCOM glue code
|
||||
NS_DECL_ISUPPORTS
|
||||
NS_DECL_NSISPEECHRECOGNITIONSERVICE
|
||||
|
||||
// Add nsIObserver code
|
||||
NS_DECL_NSIOBSERVER
|
||||
|
||||
/**
|
||||
* Default constructs a PocketSphinxSpeechRecognitionService loading default
|
||||
* files
|
||||
*/
|
||||
PocketSphinxSpeechRecognitionService();
|
||||
|
||||
private:
|
||||
/**
|
||||
* Private destructor to prevent bypassing of reference counting
|
||||
*/
|
||||
virtual ~PocketSphinxSpeechRecognitionService();
|
||||
|
||||
/** The associated SpeechRecognition */
|
||||
WeakPtr<dom::SpeechRecognition> mRecognition;
|
||||
|
||||
/**
|
||||
* Builds a mock SpeechRecognitionResultList
|
||||
*/
|
||||
dom::SpeechRecognitionResultList* BuildMockResultList();
|
||||
|
||||
/** Speex state */
|
||||
SpeexResamplerState* mSpeexState;
|
||||
|
||||
/** Pocksphix decoder */
|
||||
ps_decoder_t* mPSHandle;
|
||||
|
||||
/** Sphinxbase parsed command line arguments */
|
||||
cmd_ln_t* mPSConfig;
|
||||
|
||||
/** Flag to verify if decoder was created */
|
||||
bool ISDecoderCreated;
|
||||
|
||||
/** Flag to verify if grammar was compiled */
|
||||
bool ISGrammarCompiled;
|
||||
|
||||
/** Audio data */
|
||||
nsTArray<int16_t> mAudioVector;
|
||||
};
|
||||
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
81
dom/media/webspeech/recognition/SpeechGrammar.cpp
Normal file
81
dom/media/webspeech/recognition/SpeechGrammar.cpp
Normal file
|
|
@ -0,0 +1,81 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "SpeechGrammar.h"
|
||||
|
||||
#include "mozilla/dom/SpeechGrammarBinding.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechGrammar, mParent)
|
||||
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechGrammar)
|
||||
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechGrammar)
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechGrammar)
|
||||
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
|
||||
NS_INTERFACE_MAP_ENTRY(nsISupports)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
SpeechGrammar::SpeechGrammar(nsISupports* aParent)
|
||||
: mParent(aParent)
|
||||
{
|
||||
}
|
||||
|
||||
SpeechGrammar::~SpeechGrammar()
|
||||
{
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechGrammar>
|
||||
SpeechGrammar::Constructor(const GlobalObject& aGlobal,
|
||||
ErrorResult& aRv)
|
||||
{
|
||||
RefPtr<SpeechGrammar> speechGrammar =
|
||||
new SpeechGrammar(aGlobal.GetAsSupports());
|
||||
return speechGrammar.forget();
|
||||
}
|
||||
|
||||
nsISupports*
|
||||
SpeechGrammar::GetParentObject() const
|
||||
{
|
||||
return mParent;
|
||||
}
|
||||
|
||||
JSObject*
|
||||
SpeechGrammar::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
|
||||
{
|
||||
return SpeechGrammarBinding::Wrap(aCx, this, aGivenProto);
|
||||
}
|
||||
|
||||
void
|
||||
SpeechGrammar::GetSrc(nsString& aRetVal, ErrorResult& aRv) const
|
||||
{
|
||||
aRetVal = mSrc;
|
||||
return;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechGrammar::SetSrc(const nsAString& aArg, ErrorResult& aRv)
|
||||
{
|
||||
mSrc = aArg;
|
||||
return;
|
||||
}
|
||||
|
||||
float
|
||||
SpeechGrammar::GetWeight(ErrorResult& aRv) const
|
||||
{
|
||||
aRv.Throw(NS_ERROR_NOT_IMPLEMENTED);
|
||||
return 0;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechGrammar::SetWeight(float aArg, ErrorResult& aRv)
|
||||
{
|
||||
aRv.Throw(NS_ERROR_NOT_IMPLEMENTED);
|
||||
return;
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
59
dom/media/webspeech/recognition/SpeechGrammar.h
Normal file
59
dom/media/webspeech/recognition/SpeechGrammar.h
Normal file
|
|
@ -0,0 +1,59 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechGrammar_h
|
||||
#define mozilla_dom_SpeechGrammar_h
|
||||
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsCycleCollectionParticipant.h"
|
||||
#include "nsString.h"
|
||||
#include "nsWrapperCache.h"
|
||||
#include "js/TypeDecls.h"
|
||||
|
||||
#include "mozilla/Attributes.h"
|
||||
#include "mozilla/ErrorResult.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class GlobalObject;
|
||||
|
||||
class SpeechGrammar final : public nsISupports,
|
||||
public nsWrapperCache
|
||||
{
|
||||
public:
|
||||
explicit SpeechGrammar(nsISupports* aParent);
|
||||
|
||||
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
|
||||
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechGrammar)
|
||||
|
||||
nsISupports* GetParentObject() const;
|
||||
|
||||
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
|
||||
|
||||
static already_AddRefed<SpeechGrammar>
|
||||
Constructor(const GlobalObject& aGlobal, ErrorResult& aRv);
|
||||
|
||||
void GetSrc(nsString& aRetVal, ErrorResult& aRv) const;
|
||||
|
||||
void SetSrc(const nsAString& aArg, ErrorResult& aRv);
|
||||
|
||||
float GetWeight(ErrorResult& aRv) const;
|
||||
|
||||
void SetWeight(float aArg, ErrorResult& aRv);
|
||||
|
||||
private:
|
||||
~SpeechGrammar();
|
||||
|
||||
nsCOMPtr<nsISupports> mParent;
|
||||
|
||||
nsString mSrc;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
103
dom/media/webspeech/recognition/SpeechGrammarList.cpp
Normal file
103
dom/media/webspeech/recognition/SpeechGrammarList.cpp
Normal file
|
|
@ -0,0 +1,103 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "SpeechGrammarList.h"
|
||||
|
||||
#include "mozilla/dom/SpeechGrammarListBinding.h"
|
||||
#include "mozilla/ErrorResult.h"
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsXPCOMStrings.h"
|
||||
#include "SpeechRecognition.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechGrammarList, mParent, mItems)
|
||||
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechGrammarList)
|
||||
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechGrammarList)
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechGrammarList)
|
||||
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
|
||||
NS_INTERFACE_MAP_ENTRY(nsISupports)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
SpeechGrammarList::SpeechGrammarList(nsISupports* aParent)
|
||||
: mParent(aParent)
|
||||
{
|
||||
}
|
||||
|
||||
SpeechGrammarList::~SpeechGrammarList()
|
||||
{
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechGrammarList>
|
||||
SpeechGrammarList::Constructor(const GlobalObject& aGlobal,
|
||||
ErrorResult& aRv)
|
||||
{
|
||||
RefPtr<SpeechGrammarList> speechGrammarList =
|
||||
new SpeechGrammarList(aGlobal.GetAsSupports());
|
||||
return speechGrammarList.forget();
|
||||
}
|
||||
|
||||
JSObject*
|
||||
SpeechGrammarList::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
|
||||
{
|
||||
return SpeechGrammarListBinding::Wrap(aCx, this, aGivenProto);
|
||||
}
|
||||
|
||||
nsISupports*
|
||||
SpeechGrammarList::GetParentObject() const
|
||||
{
|
||||
return mParent;
|
||||
}
|
||||
|
||||
uint32_t
|
||||
SpeechGrammarList::Length() const
|
||||
{
|
||||
return mItems.Length();
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechGrammar>
|
||||
SpeechGrammarList::Item(uint32_t aIndex, ErrorResult& aRv)
|
||||
{
|
||||
RefPtr<SpeechGrammar> result = mItems.ElementAt(aIndex);
|
||||
return result.forget();
|
||||
}
|
||||
|
||||
void
|
||||
SpeechGrammarList::AddFromURI(const nsAString& aSrc,
|
||||
const Optional<float>& aWeight,
|
||||
ErrorResult& aRv)
|
||||
{
|
||||
aRv.Throw(NS_ERROR_NOT_IMPLEMENTED);
|
||||
return;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechGrammarList::AddFromString(const nsAString& aString,
|
||||
const Optional<float>& aWeight,
|
||||
ErrorResult& aRv)
|
||||
{
|
||||
SpeechGrammar* speechGrammar = new SpeechGrammar(mParent);
|
||||
speechGrammar->SetSrc(aString, aRv);
|
||||
mItems.AppendElement(speechGrammar);
|
||||
return;
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechGrammar>
|
||||
SpeechGrammarList::IndexedGetter(uint32_t aIndex, bool& aPresent,
|
||||
ErrorResult& aRv)
|
||||
{
|
||||
if (aIndex >= Length()) {
|
||||
aPresent = false;
|
||||
return nullptr;
|
||||
}
|
||||
ErrorResult rv;
|
||||
aPresent = true;
|
||||
return Item(aIndex, rv);
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
63
dom/media/webspeech/recognition/SpeechGrammarList.h
Normal file
63
dom/media/webspeech/recognition/SpeechGrammarList.h
Normal file
|
|
@ -0,0 +1,63 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechGrammarList_h
|
||||
#define mozilla_dom_SpeechGrammarList_h
|
||||
|
||||
#include "mozilla/Attributes.h"
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsCycleCollectionParticipant.h"
|
||||
#include "nsWrapperCache.h"
|
||||
|
||||
struct JSContext;
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
class ErrorResult;
|
||||
|
||||
namespace dom {
|
||||
|
||||
class GlobalObject;
|
||||
class SpeechGrammar;
|
||||
template<typename> class Optional;
|
||||
|
||||
class SpeechGrammarList final : public nsISupports,
|
||||
public nsWrapperCache
|
||||
{
|
||||
public:
|
||||
explicit SpeechGrammarList(nsISupports* aParent);
|
||||
|
||||
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
|
||||
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechGrammarList)
|
||||
|
||||
static already_AddRefed<SpeechGrammarList> Constructor(const GlobalObject& aGlobal, ErrorResult& aRv);
|
||||
|
||||
nsISupports* GetParentObject() const;
|
||||
|
||||
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
|
||||
|
||||
uint32_t Length() const;
|
||||
|
||||
already_AddRefed<SpeechGrammar> Item(uint32_t aIndex, ErrorResult& aRv);
|
||||
|
||||
void AddFromURI(const nsAString& aSrc, const Optional<float>& aWeight, ErrorResult& aRv);
|
||||
|
||||
void AddFromString(const nsAString& aString, const Optional<float>& aWeight, ErrorResult& aRv);
|
||||
|
||||
already_AddRefed<SpeechGrammar> IndexedGetter(uint32_t aIndex, bool& aPresent, ErrorResult& aRv);
|
||||
|
||||
private:
|
||||
~SpeechGrammarList();
|
||||
|
||||
nsCOMPtr<nsISupports> mParent;
|
||||
|
||||
nsTArray<RefPtr<SpeechGrammar>> mItems;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
1087
dom/media/webspeech/recognition/SpeechRecognition.cpp
Normal file
1087
dom/media/webspeech/recognition/SpeechRecognition.cpp
Normal file
File diff suppressed because it is too large
Load diff
296
dom/media/webspeech/recognition/SpeechRecognition.h
Normal file
296
dom/media/webspeech/recognition/SpeechRecognition.h
Normal file
|
|
@ -0,0 +1,296 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechRecognition_h
|
||||
#define mozilla_dom_SpeechRecognition_h
|
||||
|
||||
#include "mozilla/Attributes.h"
|
||||
#include "mozilla/DOMEventTargetHelper.h"
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsString.h"
|
||||
#include "nsWrapperCache.h"
|
||||
#include "nsTArray.h"
|
||||
#include "js/TypeDecls.h"
|
||||
|
||||
#include "nsIDOMNavigatorUserMedia.h"
|
||||
#include "nsITimer.h"
|
||||
#include "MediaEngine.h"
|
||||
#include "MediaStreamGraph.h"
|
||||
#include "AudioSegment.h"
|
||||
#include "mozilla/WeakPtr.h"
|
||||
|
||||
#include "SpeechGrammarList.h"
|
||||
#include "SpeechRecognitionResultList.h"
|
||||
#include "SpeechStreamListener.h"
|
||||
#include "nsISpeechRecognitionService.h"
|
||||
#include "endpointer.h"
|
||||
|
||||
#include "mozilla/dom/SpeechRecognitionError.h"
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
namespace dom {
|
||||
|
||||
#define SPEECH_RECOGNITION_TEST_EVENT_REQUEST_TOPIC "SpeechRecognitionTest:RequestEvent"
|
||||
#define SPEECH_RECOGNITION_TEST_END_TOPIC "SpeechRecognitionTest:End"
|
||||
|
||||
class GlobalObject;
|
||||
class SpeechEvent;
|
||||
|
||||
LogModule* GetSpeechRecognitionLog();
|
||||
#define SR_LOG(...) MOZ_LOG(GetSpeechRecognitionLog(), mozilla::LogLevel::Debug, (__VA_ARGS__))
|
||||
|
||||
class SpeechRecognition final : public DOMEventTargetHelper,
|
||||
public nsIObserver,
|
||||
public SupportsWeakPtr<SpeechRecognition>
|
||||
{
|
||||
public:
|
||||
MOZ_DECLARE_WEAKREFERENCE_TYPENAME(SpeechRecognition)
|
||||
explicit SpeechRecognition(nsPIDOMWindowInner* aOwnerWindow);
|
||||
|
||||
NS_DECL_ISUPPORTS_INHERITED
|
||||
NS_DECL_CYCLE_COLLECTION_CLASS_INHERITED(SpeechRecognition, DOMEventTargetHelper)
|
||||
|
||||
NS_DECL_NSIOBSERVER
|
||||
|
||||
nsISupports* GetParentObject() const;
|
||||
|
||||
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
|
||||
|
||||
static bool IsAuthorized(JSContext* aCx, JSObject* aGlobal);
|
||||
|
||||
static already_AddRefed<SpeechRecognition>
|
||||
Constructor(const GlobalObject& aGlobal, ErrorResult& aRv);
|
||||
|
||||
already_AddRefed<SpeechGrammarList> Grammars() const;
|
||||
|
||||
void SetGrammars(mozilla::dom::SpeechGrammarList& aArg);
|
||||
|
||||
void GetLang(nsString& aRetVal) const;
|
||||
|
||||
void SetLang(const nsAString& aArg);
|
||||
|
||||
bool GetContinuous(ErrorResult& aRv) const;
|
||||
|
||||
void SetContinuous(bool aArg, ErrorResult& aRv);
|
||||
|
||||
bool InterimResults() const;
|
||||
|
||||
void SetInterimResults(bool aArg);
|
||||
|
||||
uint32_t MaxAlternatives() const;
|
||||
|
||||
void SetMaxAlternatives(uint32_t aArg);
|
||||
|
||||
void GetServiceURI(nsString& aRetVal, ErrorResult& aRv) const;
|
||||
|
||||
void SetServiceURI(const nsAString& aArg, ErrorResult& aRv);
|
||||
|
||||
void Start(const Optional<NonNull<DOMMediaStream>>& aStream, ErrorResult& aRv);
|
||||
|
||||
void Stop();
|
||||
|
||||
void Abort();
|
||||
|
||||
IMPL_EVENT_HANDLER(audiostart)
|
||||
IMPL_EVENT_HANDLER(soundstart)
|
||||
IMPL_EVENT_HANDLER(speechstart)
|
||||
IMPL_EVENT_HANDLER(speechend)
|
||||
IMPL_EVENT_HANDLER(soundend)
|
||||
IMPL_EVENT_HANDLER(audioend)
|
||||
IMPL_EVENT_HANDLER(result)
|
||||
IMPL_EVENT_HANDLER(nomatch)
|
||||
IMPL_EVENT_HANDLER(error)
|
||||
IMPL_EVENT_HANDLER(start)
|
||||
IMPL_EVENT_HANDLER(end)
|
||||
|
||||
enum EventType {
|
||||
EVENT_START,
|
||||
EVENT_STOP,
|
||||
EVENT_ABORT,
|
||||
EVENT_AUDIO_DATA,
|
||||
EVENT_AUDIO_ERROR,
|
||||
EVENT_RECOGNITIONSERVICE_INTERMEDIATE_RESULT,
|
||||
EVENT_RECOGNITIONSERVICE_FINAL_RESULT,
|
||||
EVENT_RECOGNITIONSERVICE_ERROR,
|
||||
EVENT_COUNT
|
||||
};
|
||||
|
||||
void DispatchError(EventType aErrorType, SpeechRecognitionErrorCode aErrorCode, const nsAString& aMessage);
|
||||
uint32_t FillSamplesBuffer(const int16_t* aSamples, uint32_t aSampleCount);
|
||||
uint32_t SplitSamplesBuffer(const int16_t* aSamplesBuffer, uint32_t aSampleCount, nsTArray<RefPtr<SharedBuffer>>& aResult);
|
||||
AudioSegment* CreateAudioSegment(nsTArray<RefPtr<SharedBuffer>>& aChunks);
|
||||
void FeedAudioData(already_AddRefed<SharedBuffer> aSamples, uint32_t aDuration, MediaStreamListener* aProvider, TrackRate aTrackRate);
|
||||
|
||||
friend class SpeechEvent;
|
||||
private:
|
||||
virtual ~SpeechRecognition() {};
|
||||
|
||||
enum FSMState {
|
||||
STATE_IDLE,
|
||||
STATE_STARTING,
|
||||
STATE_ESTIMATING,
|
||||
STATE_WAITING_FOR_SPEECH,
|
||||
STATE_RECOGNIZING,
|
||||
STATE_WAITING_FOR_RESULT,
|
||||
STATE_COUNT
|
||||
};
|
||||
|
||||
void SetState(FSMState state);
|
||||
bool StateBetween(FSMState begin, FSMState end);
|
||||
|
||||
bool SetRecognitionService(ErrorResult& aRv);
|
||||
bool ValidateAndSetGrammarList(ErrorResult& aRv);
|
||||
|
||||
class GetUserMediaSuccessCallback : public nsIDOMGetUserMediaSuccessCallback
|
||||
{
|
||||
public:
|
||||
NS_DECL_ISUPPORTS
|
||||
NS_DECL_NSIDOMGETUSERMEDIASUCCESSCALLBACK
|
||||
|
||||
explicit GetUserMediaSuccessCallback(SpeechRecognition* aRecognition)
|
||||
: mRecognition(aRecognition)
|
||||
{}
|
||||
|
||||
private:
|
||||
virtual ~GetUserMediaSuccessCallback() {}
|
||||
|
||||
RefPtr<SpeechRecognition> mRecognition;
|
||||
};
|
||||
|
||||
class GetUserMediaErrorCallback : public nsIDOMGetUserMediaErrorCallback
|
||||
{
|
||||
public:
|
||||
NS_DECL_ISUPPORTS
|
||||
NS_DECL_NSIDOMGETUSERMEDIAERRORCALLBACK
|
||||
|
||||
explicit GetUserMediaErrorCallback(SpeechRecognition* aRecognition)
|
||||
: mRecognition(aRecognition)
|
||||
{}
|
||||
|
||||
private:
|
||||
virtual ~GetUserMediaErrorCallback() {}
|
||||
|
||||
RefPtr<SpeechRecognition> mRecognition;
|
||||
};
|
||||
|
||||
NS_IMETHOD StartRecording(DOMMediaStream* aDOMStream);
|
||||
NS_IMETHOD StopRecording();
|
||||
|
||||
uint32_t ProcessAudioSegment(AudioSegment* aSegment, TrackRate aTrackRate);
|
||||
void NotifyError(SpeechEvent* aEvent);
|
||||
|
||||
void ProcessEvent(SpeechEvent* aEvent);
|
||||
void Transition(SpeechEvent* aEvent);
|
||||
|
||||
void Reset();
|
||||
void ResetAndEnd();
|
||||
void WaitForAudioData(SpeechEvent* aEvent);
|
||||
void StartedAudioCapture(SpeechEvent* aEvent);
|
||||
void StopRecordingAndRecognize(SpeechEvent* aEvent);
|
||||
void WaitForEstimation(SpeechEvent* aEvent);
|
||||
void DetectSpeech(SpeechEvent* aEvent);
|
||||
void WaitForSpeechEnd(SpeechEvent* aEvent);
|
||||
void NotifyFinalResult(SpeechEvent* aEvent);
|
||||
void DoNothing(SpeechEvent* aEvent);
|
||||
void AbortSilently(SpeechEvent* aEvent);
|
||||
void AbortError(SpeechEvent* aEvent);
|
||||
|
||||
RefPtr<DOMMediaStream> mDOMStream;
|
||||
RefPtr<SpeechStreamListener> mSpeechListener;
|
||||
nsCOMPtr<nsISpeechRecognitionService> mRecognitionService;
|
||||
|
||||
FSMState mCurrentState;
|
||||
|
||||
Endpointer mEndpointer;
|
||||
uint32_t mEstimationSamples;
|
||||
|
||||
uint32_t mAudioSamplesPerChunk;
|
||||
|
||||
// buffer holds one chunk of mAudioSamplesPerChunk
|
||||
// samples before feeding it to mEndpointer
|
||||
RefPtr<SharedBuffer> mAudioSamplesBuffer;
|
||||
uint32_t mBufferedSamples;
|
||||
|
||||
nsCOMPtr<nsITimer> mSpeechDetectionTimer;
|
||||
bool mAborted;
|
||||
|
||||
nsString mLang;
|
||||
|
||||
RefPtr<SpeechGrammarList> mSpeechGrammarList;
|
||||
|
||||
// WebSpeechAPI (http://bit.ly/1gIl7DC) states:
|
||||
//
|
||||
// 1. Default value MUST be false
|
||||
// 2. If true, interim results SHOULD be returned
|
||||
// 3. If false, interim results MUST NOT be returned
|
||||
//
|
||||
// Pocketsphinx does not return interm results; so, defaulting
|
||||
// mInterimResults to false, then ignoring its subsequent value
|
||||
// is a conforming implementation.
|
||||
bool mInterimResults;
|
||||
|
||||
// WebSpeechAPI (http://bit.ly/1JAiqeo) states:
|
||||
//
|
||||
// 1. Default value is 1
|
||||
// 2. Subsequent value is the "maximum number of SpeechRecognitionAlternatives per result"
|
||||
//
|
||||
// Pocketsphinx can only return at maximum a single SpeechRecognitionAlternative
|
||||
// per SpeechRecognitionResult. So defaulting mMaxAlternatives to 1, for all non
|
||||
// zero values ignoring mMaxAlternatives while for a 0 value returning no
|
||||
// SpeechRecognitionAlternative per result is a conforming implementation.
|
||||
uint32_t mMaxAlternatives;
|
||||
|
||||
void ProcessTestEventRequest(nsISupports* aSubject, const nsAString& aEventName);
|
||||
|
||||
const char* GetName(FSMState aId);
|
||||
const char* GetName(SpeechEvent* aId);
|
||||
};
|
||||
|
||||
class SpeechEvent : public Runnable
|
||||
{
|
||||
public:
|
||||
SpeechEvent(SpeechRecognition* aRecognition, SpeechRecognition::EventType aType)
|
||||
: mAudioSegment(0)
|
||||
, mRecognitionResultList(nullptr)
|
||||
, mError(nullptr)
|
||||
, mRecognition(aRecognition)
|
||||
, mType(aType)
|
||||
, mTrackRate(0)
|
||||
{
|
||||
}
|
||||
|
||||
~SpeechEvent();
|
||||
|
||||
NS_IMETHOD Run() override;
|
||||
AudioSegment* mAudioSegment;
|
||||
RefPtr<SpeechRecognitionResultList> mRecognitionResultList; // TODO: make this a session being passed which also has index and stuff
|
||||
RefPtr<SpeechRecognitionError> mError;
|
||||
|
||||
friend class SpeechRecognition;
|
||||
private:
|
||||
SpeechRecognition* mRecognition;
|
||||
|
||||
// for AUDIO_DATA events, keep a reference to the provider
|
||||
// of the data (i.e., the SpeechStreamListener) to ensure it
|
||||
// is kept alive (and keeps SpeechRecognition alive) until this
|
||||
// event gets processed.
|
||||
RefPtr<MediaStreamListener> mProvider;
|
||||
SpeechRecognition::EventType mType;
|
||||
TrackRate mTrackRate;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
|
||||
inline nsISupports*
|
||||
ToSupports(dom::SpeechRecognition* aRec)
|
||||
{
|
||||
return ToSupports(static_cast<DOMEventTargetHelper*>(aRec));
|
||||
}
|
||||
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
|
|
@ -0,0 +1,59 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "SpeechRecognitionAlternative.h"
|
||||
|
||||
#include "mozilla/dom/SpeechRecognitionAlternativeBinding.h"
|
||||
|
||||
#include "SpeechRecognition.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechRecognitionAlternative, mParent)
|
||||
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechRecognitionAlternative)
|
||||
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechRecognitionAlternative)
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechRecognitionAlternative)
|
||||
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
|
||||
NS_INTERFACE_MAP_ENTRY(nsISupports)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
SpeechRecognitionAlternative::SpeechRecognitionAlternative(SpeechRecognition* aParent)
|
||||
: mConfidence(0)
|
||||
, mParent(aParent)
|
||||
{
|
||||
}
|
||||
|
||||
SpeechRecognitionAlternative::~SpeechRecognitionAlternative()
|
||||
{
|
||||
}
|
||||
|
||||
JSObject*
|
||||
SpeechRecognitionAlternative::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
|
||||
{
|
||||
return SpeechRecognitionAlternativeBinding::Wrap(aCx, this, aGivenProto);
|
||||
}
|
||||
|
||||
nsISupports*
|
||||
SpeechRecognitionAlternative::GetParentObject() const
|
||||
{
|
||||
return static_cast<DOMEventTargetHelper*>(mParent.get());
|
||||
}
|
||||
|
||||
void
|
||||
SpeechRecognitionAlternative::GetTranscript(nsString& aRetVal) const
|
||||
{
|
||||
aRetVal = mTranscript;
|
||||
}
|
||||
|
||||
float
|
||||
SpeechRecognitionAlternative::Confidence() const
|
||||
{
|
||||
return mConfidence;
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
|
@ -0,0 +1,50 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechRecognitionAlternative_h
|
||||
#define mozilla_dom_SpeechRecognitionAlternative_h
|
||||
|
||||
#include "nsCycleCollectionParticipant.h"
|
||||
#include "nsString.h"
|
||||
#include "nsWrapperCache.h"
|
||||
#include "js/TypeDecls.h"
|
||||
|
||||
#include "mozilla/Attributes.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class SpeechRecognition;
|
||||
|
||||
class SpeechRecognitionAlternative final : public nsISupports,
|
||||
public nsWrapperCache
|
||||
{
|
||||
public:
|
||||
explicit SpeechRecognitionAlternative(SpeechRecognition* aParent);
|
||||
|
||||
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
|
||||
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechRecognitionAlternative)
|
||||
|
||||
nsISupports* GetParentObject() const;
|
||||
|
||||
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
|
||||
|
||||
void GetTranscript(nsString& aRetVal) const;
|
||||
|
||||
float Confidence() const;
|
||||
|
||||
nsString mTranscript;
|
||||
float mConfidence;
|
||||
private:
|
||||
~SpeechRecognitionAlternative();
|
||||
|
||||
RefPtr<SpeechRecognition> mParent;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
76
dom/media/webspeech/recognition/SpeechRecognitionResult.cpp
Normal file
76
dom/media/webspeech/recognition/SpeechRecognitionResult.cpp
Normal file
|
|
@ -0,0 +1,76 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "SpeechRecognitionResult.h"
|
||||
#include "mozilla/dom/SpeechRecognitionResultBinding.h"
|
||||
|
||||
#include "SpeechRecognition.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechRecognitionResult, mParent)
|
||||
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechRecognitionResult)
|
||||
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechRecognitionResult)
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechRecognitionResult)
|
||||
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
|
||||
NS_INTERFACE_MAP_ENTRY(nsISupports)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
SpeechRecognitionResult::SpeechRecognitionResult(SpeechRecognition* aParent)
|
||||
: mParent(aParent)
|
||||
{
|
||||
}
|
||||
|
||||
SpeechRecognitionResult::~SpeechRecognitionResult()
|
||||
{
|
||||
}
|
||||
|
||||
JSObject*
|
||||
SpeechRecognitionResult::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
|
||||
{
|
||||
return SpeechRecognitionResultBinding::Wrap(aCx, this, aGivenProto);
|
||||
}
|
||||
|
||||
nsISupports*
|
||||
SpeechRecognitionResult::GetParentObject() const
|
||||
{
|
||||
return static_cast<DOMEventTargetHelper*>(mParent.get());
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechRecognitionAlternative>
|
||||
SpeechRecognitionResult::IndexedGetter(uint32_t aIndex, bool& aPresent)
|
||||
{
|
||||
if (aIndex >= Length()) {
|
||||
aPresent = false;
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
aPresent = true;
|
||||
return Item(aIndex);
|
||||
}
|
||||
|
||||
uint32_t
|
||||
SpeechRecognitionResult::Length() const
|
||||
{
|
||||
return mItems.Length();
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechRecognitionAlternative>
|
||||
SpeechRecognitionResult::Item(uint32_t aIndex)
|
||||
{
|
||||
RefPtr<SpeechRecognitionAlternative> alternative = mItems.ElementAt(aIndex);
|
||||
return alternative.forget();
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechRecognitionResult::IsFinal() const
|
||||
{
|
||||
return true; // TODO
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
55
dom/media/webspeech/recognition/SpeechRecognitionResult.h
Normal file
55
dom/media/webspeech/recognition/SpeechRecognitionResult.h
Normal file
|
|
@ -0,0 +1,55 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechRecognitionResult_h
|
||||
#define mozilla_dom_SpeechRecognitionResult_h
|
||||
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsCycleCollectionParticipant.h"
|
||||
#include "nsWrapperCache.h"
|
||||
#include "nsTArray.h"
|
||||
#include "js/TypeDecls.h"
|
||||
|
||||
#include "mozilla/Attributes.h"
|
||||
|
||||
#include "SpeechRecognitionAlternative.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class SpeechRecognitionResult final : public nsISupports,
|
||||
public nsWrapperCache
|
||||
{
|
||||
public:
|
||||
explicit SpeechRecognitionResult(SpeechRecognition* aParent);
|
||||
|
||||
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
|
||||
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechRecognitionResult)
|
||||
|
||||
nsISupports* GetParentObject() const;
|
||||
|
||||
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
|
||||
|
||||
uint32_t Length() const;
|
||||
|
||||
already_AddRefed<SpeechRecognitionAlternative> Item(uint32_t aIndex);
|
||||
|
||||
bool IsFinal() const;
|
||||
|
||||
already_AddRefed<SpeechRecognitionAlternative> IndexedGetter(uint32_t aIndex, bool& aPresent);
|
||||
|
||||
nsTArray<RefPtr<SpeechRecognitionAlternative>> mItems;
|
||||
|
||||
private:
|
||||
~SpeechRecognitionResult();
|
||||
|
||||
RefPtr<SpeechRecognition> mParent;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
|
|
@ -0,0 +1,71 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "SpeechRecognitionResultList.h"
|
||||
|
||||
#include "mozilla/dom/SpeechRecognitionResultListBinding.h"
|
||||
|
||||
#include "SpeechRecognition.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechRecognitionResultList, mParent, mItems)
|
||||
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechRecognitionResultList)
|
||||
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechRecognitionResultList)
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechRecognitionResultList)
|
||||
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
|
||||
NS_INTERFACE_MAP_ENTRY(nsISupports)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
SpeechRecognitionResultList::SpeechRecognitionResultList(SpeechRecognition* aParent)
|
||||
: mParent(aParent)
|
||||
{
|
||||
}
|
||||
|
||||
SpeechRecognitionResultList::~SpeechRecognitionResultList()
|
||||
{
|
||||
}
|
||||
|
||||
nsISupports*
|
||||
SpeechRecognitionResultList::GetParentObject() const
|
||||
{
|
||||
return static_cast<DOMEventTargetHelper*>(mParent.get());
|
||||
}
|
||||
|
||||
JSObject*
|
||||
SpeechRecognitionResultList::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
|
||||
{
|
||||
return SpeechRecognitionResultListBinding::Wrap(aCx, this, aGivenProto);
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechRecognitionResult>
|
||||
SpeechRecognitionResultList::IndexedGetter(uint32_t aIndex, bool& aPresent)
|
||||
{
|
||||
if (aIndex >= Length()) {
|
||||
aPresent = false;
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
aPresent = true;
|
||||
return Item(aIndex);
|
||||
}
|
||||
|
||||
uint32_t
|
||||
SpeechRecognitionResultList::Length() const
|
||||
{
|
||||
return mItems.Length();
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechRecognitionResult>
|
||||
SpeechRecognitionResultList::Item(uint32_t aIndex)
|
||||
{
|
||||
RefPtr<SpeechRecognitionResult> result = mItems.ElementAt(aIndex);
|
||||
return result.forget();
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
|
@ -0,0 +1,53 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechRecognitionResultList_h
|
||||
#define mozilla_dom_SpeechRecognitionResultList_h
|
||||
|
||||
#include "nsCycleCollectionParticipant.h"
|
||||
#include "nsWrapperCache.h"
|
||||
#include "nsTArray.h"
|
||||
#include "js/TypeDecls.h"
|
||||
|
||||
#include "mozilla/Attributes.h"
|
||||
|
||||
#include "SpeechRecognitionResult.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class SpeechRecognition;
|
||||
|
||||
class SpeechRecognitionResultList final : public nsISupports,
|
||||
public nsWrapperCache
|
||||
{
|
||||
public:
|
||||
explicit SpeechRecognitionResultList(SpeechRecognition* aParent);
|
||||
|
||||
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
|
||||
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechRecognitionResultList)
|
||||
|
||||
nsISupports* GetParentObject() const;
|
||||
|
||||
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
|
||||
|
||||
uint32_t Length() const;
|
||||
|
||||
already_AddRefed<SpeechRecognitionResult> Item(uint32_t aIndex);
|
||||
|
||||
already_AddRefed<SpeechRecognitionResult> IndexedGetter(uint32_t aIndex, bool& aPresent);
|
||||
|
||||
nsTArray<RefPtr<SpeechRecognitionResult>> mItems;
|
||||
private:
|
||||
~SpeechRecognitionResultList();
|
||||
|
||||
RefPtr<SpeechRecognition> mParent;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
94
dom/media/webspeech/recognition/SpeechStreamListener.cpp
Normal file
94
dom/media/webspeech/recognition/SpeechStreamListener.cpp
Normal file
|
|
@ -0,0 +1,94 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "SpeechStreamListener.h"
|
||||
|
||||
#include "SpeechRecognition.h"
|
||||
#include "nsProxyRelease.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
SpeechStreamListener::SpeechStreamListener(SpeechRecognition* aRecognition)
|
||||
: mRecognition(aRecognition)
|
||||
{
|
||||
}
|
||||
|
||||
SpeechStreamListener::~SpeechStreamListener()
|
||||
{
|
||||
nsCOMPtr<nsIThread> mainThread;
|
||||
NS_GetMainThread(getter_AddRefs(mainThread));
|
||||
|
||||
NS_ProxyRelease(mainThread, mRecognition.forget());
|
||||
}
|
||||
|
||||
void
|
||||
SpeechStreamListener::NotifyQueuedAudioData(MediaStreamGraph* aGraph, TrackID aID,
|
||||
StreamTime aTrackOffset,
|
||||
const AudioSegment& aQueuedMedia,
|
||||
MediaStream* aInputStream,
|
||||
TrackID aInputTrackID)
|
||||
{
|
||||
AudioSegment* audio = const_cast<AudioSegment*>(
|
||||
static_cast<const AudioSegment*>(&aQueuedMedia));
|
||||
|
||||
AudioSegment::ChunkIterator iterator(*audio);
|
||||
while (!iterator.IsEnded()) {
|
||||
// Skip over-large chunks so we don't crash!
|
||||
if (iterator->GetDuration() > INT_MAX) {
|
||||
continue;
|
||||
}
|
||||
int duration = int(iterator->GetDuration());
|
||||
|
||||
if (iterator->IsNull()) {
|
||||
nsTArray<int16_t> nullData;
|
||||
PodZero(nullData.AppendElements(duration), duration);
|
||||
ConvertAndDispatchAudioChunk(duration, iterator->mVolume,
|
||||
nullData.Elements(), aGraph->GraphRate());
|
||||
} else {
|
||||
AudioSampleFormat format = iterator->mBufferFormat;
|
||||
|
||||
MOZ_ASSERT(format == AUDIO_FORMAT_S16 || format == AUDIO_FORMAT_FLOAT32);
|
||||
|
||||
if (format == AUDIO_FORMAT_S16) {
|
||||
ConvertAndDispatchAudioChunk(duration,iterator->mVolume,
|
||||
static_cast<const int16_t*>(iterator->mChannelData[0]),
|
||||
aGraph->GraphRate());
|
||||
} else if (format == AUDIO_FORMAT_FLOAT32) {
|
||||
ConvertAndDispatchAudioChunk(duration,iterator->mVolume,
|
||||
static_cast<const float*>(iterator->mChannelData[0]),
|
||||
aGraph->GraphRate());
|
||||
}
|
||||
}
|
||||
|
||||
iterator.Next();
|
||||
}
|
||||
}
|
||||
|
||||
template<typename SampleFormatType> void
|
||||
SpeechStreamListener::ConvertAndDispatchAudioChunk(int aDuration, float aVolume,
|
||||
SampleFormatType* aData,
|
||||
TrackRate aTrackRate)
|
||||
{
|
||||
RefPtr<SharedBuffer> samples(SharedBuffer::Create(aDuration *
|
||||
1 * // channel
|
||||
sizeof(int16_t)));
|
||||
|
||||
int16_t* to = static_cast<int16_t*>(samples->Data());
|
||||
ConvertAudioSamplesWithScale(aData, to, aDuration, aVolume);
|
||||
|
||||
mRecognition->FeedAudioData(samples.forget(), aDuration, this, aTrackRate);
|
||||
}
|
||||
|
||||
void
|
||||
SpeechStreamListener::NotifyEvent(MediaStreamGraph* aGraph,
|
||||
MediaStreamGraphEvent event)
|
||||
{
|
||||
// TODO dispatch SpeechEnd event so services can be informed
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
46
dom/media/webspeech/recognition/SpeechStreamListener.h
Normal file
46
dom/media/webspeech/recognition/SpeechStreamListener.h
Normal file
|
|
@ -0,0 +1,46 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechStreamListener_h
|
||||
#define mozilla_dom_SpeechStreamListener_h
|
||||
|
||||
#include "MediaStreamGraph.h"
|
||||
#include "MediaStreamListener.h"
|
||||
#include "AudioSegment.h"
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
class AudioSegment;
|
||||
|
||||
namespace dom {
|
||||
|
||||
class SpeechRecognition;
|
||||
|
||||
class SpeechStreamListener : public MediaStreamListener
|
||||
{
|
||||
public:
|
||||
explicit SpeechStreamListener(SpeechRecognition* aRecognition);
|
||||
~SpeechStreamListener();
|
||||
|
||||
void NotifyQueuedAudioData(MediaStreamGraph* aGraph, TrackID aID,
|
||||
StreamTime aTrackOffset,
|
||||
const AudioSegment& aQueuedMedia,
|
||||
MediaStream* aInputStream,
|
||||
TrackID aInputTrackID) override;
|
||||
|
||||
void NotifyEvent(MediaStreamGraph* aGraph,
|
||||
MediaStreamGraphEvent event) override;
|
||||
|
||||
private:
|
||||
template<typename SampleFormatType>
|
||||
void ConvertAndDispatchAudioChunk(int aDuration, float aVolume, SampleFormatType* aData, TrackRate aTrackRate);
|
||||
RefPtr<SpeechRecognition> mRecognition;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
193
dom/media/webspeech/recognition/endpointer.cc
Normal file
193
dom/media/webspeech/recognition/endpointer.cc
Normal file
|
|
@ -0,0 +1,193 @@
|
|||
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without
|
||||
// modification, are permitted provided that the following conditions are
|
||||
// met:
|
||||
//
|
||||
// * Redistributions of source code must retain the above copyright
|
||||
// notice, this list of conditions and the following disclaimer.
|
||||
// * Redistributions in binary form must reproduce the above
|
||||
// copyright notice, this list of conditions and the following disclaimer
|
||||
// in the documentation and/or other materials provided with the
|
||||
// distribution.
|
||||
// * Neither the name of Google Inc. nor the names of its
|
||||
// contributors may be used to endorse or promote products derived from
|
||||
// this software without specific prior written permission.
|
||||
//
|
||||
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
|
||||
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
|
||||
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
||||
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
|
||||
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
||||
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
||||
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
#include "endpointer.h"
|
||||
|
||||
#include "AudioSegment.h"
|
||||
|
||||
namespace {
|
||||
const int kFrameRate = 200; // 1 frame = 5ms of audio.
|
||||
}
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
Endpointer::Endpointer(int sample_rate)
|
||||
: speech_input_possibly_complete_silence_length_us_(-1),
|
||||
speech_input_complete_silence_length_us_(-1),
|
||||
audio_frame_time_us_(0),
|
||||
sample_rate_(sample_rate),
|
||||
frame_size_(0) {
|
||||
Reset();
|
||||
|
||||
frame_size_ = static_cast<int>(sample_rate / static_cast<float>(kFrameRate));
|
||||
|
||||
speech_input_minimum_length_us_ =
|
||||
static_cast<int64_t>(1.7 * 1000000);
|
||||
speech_input_complete_silence_length_us_ =
|
||||
static_cast<int64_t>(0.5 * 1000000);
|
||||
long_speech_input_complete_silence_length_us_ = -1;
|
||||
long_speech_length_us_ = -1;
|
||||
speech_input_possibly_complete_silence_length_us_ =
|
||||
1 * 1000000;
|
||||
|
||||
// Set the default configuration for Push To Talk mode.
|
||||
EnergyEndpointerParams ep_config;
|
||||
ep_config.set_frame_period(1.0f / static_cast<float>(kFrameRate));
|
||||
ep_config.set_frame_duration(1.0f / static_cast<float>(kFrameRate));
|
||||
ep_config.set_endpoint_margin(0.2f);
|
||||
ep_config.set_onset_window(0.15f);
|
||||
ep_config.set_speech_on_window(0.4f);
|
||||
ep_config.set_offset_window(0.15f);
|
||||
ep_config.set_onset_detect_dur(0.09f);
|
||||
ep_config.set_onset_confirm_dur(0.075f);
|
||||
ep_config.set_on_maintain_dur(0.10f);
|
||||
ep_config.set_offset_confirm_dur(0.12f);
|
||||
ep_config.set_decision_threshold(1000.0f);
|
||||
ep_config.set_min_decision_threshold(50.0f);
|
||||
ep_config.set_fast_update_dur(0.2f);
|
||||
ep_config.set_sample_rate(static_cast<float>(sample_rate));
|
||||
ep_config.set_min_fundamental_frequency(57.143f);
|
||||
ep_config.set_max_fundamental_frequency(400.0f);
|
||||
ep_config.set_contamination_rejection_period(0.25f);
|
||||
energy_endpointer_.Init(ep_config);
|
||||
}
|
||||
|
||||
void Endpointer::Reset() {
|
||||
old_ep_status_ = EP_PRE_SPEECH;
|
||||
waiting_for_speech_possibly_complete_timeout_ = false;
|
||||
waiting_for_speech_complete_timeout_ = false;
|
||||
speech_previously_detected_ = false;
|
||||
speech_input_complete_ = false;
|
||||
audio_frame_time_us_ = 0; // Reset time for packets sent to endpointer.
|
||||
speech_end_time_us_ = -1;
|
||||
speech_start_time_us_ = -1;
|
||||
}
|
||||
|
||||
void Endpointer::StartSession() {
|
||||
Reset();
|
||||
energy_endpointer_.StartSession();
|
||||
}
|
||||
|
||||
void Endpointer::EndSession() {
|
||||
energy_endpointer_.EndSession();
|
||||
}
|
||||
|
||||
void Endpointer::SetEnvironmentEstimationMode() {
|
||||
Reset();
|
||||
energy_endpointer_.SetEnvironmentEstimationMode();
|
||||
}
|
||||
|
||||
void Endpointer::SetUserInputMode() {
|
||||
energy_endpointer_.SetUserInputMode();
|
||||
}
|
||||
|
||||
EpStatus Endpointer::Status(int64_t *time) {
|
||||
return energy_endpointer_.Status(time);
|
||||
}
|
||||
|
||||
EpStatus Endpointer::ProcessAudio(const AudioChunk& raw_audio, float* rms_out) {
|
||||
MOZ_ASSERT(raw_audio.mBufferFormat == AUDIO_FORMAT_S16, "Audio is not in 16 bit format");
|
||||
const int16_t* audio_data = static_cast<const int16_t*>(raw_audio.mChannelData[0]);
|
||||
const int num_samples = raw_audio.mDuration;
|
||||
EpStatus ep_status = EP_PRE_SPEECH;
|
||||
|
||||
// Process the input data in blocks of frame_size_, dropping any incomplete
|
||||
// frames at the end (which is ok since typically the caller will be recording
|
||||
// audio in multiples of our frame size).
|
||||
int sample_index = 0;
|
||||
while (sample_index + frame_size_ <= num_samples) {
|
||||
// Have the endpointer process the frame.
|
||||
energy_endpointer_.ProcessAudioFrame(audio_frame_time_us_,
|
||||
audio_data + sample_index,
|
||||
frame_size_,
|
||||
rms_out);
|
||||
sample_index += frame_size_;
|
||||
audio_frame_time_us_ += (frame_size_ * 1000000) /
|
||||
sample_rate_;
|
||||
|
||||
// Get the status of the endpointer.
|
||||
int64_t ep_time;
|
||||
ep_status = energy_endpointer_.Status(&ep_time);
|
||||
if (old_ep_status_ != ep_status)
|
||||
fprintf(stderr, "Status changed old= %d, new= %d\n", old_ep_status_, ep_status);
|
||||
|
||||
// Handle state changes.
|
||||
if ((EP_SPEECH_PRESENT == ep_status) &&
|
||||
(EP_POSSIBLE_ONSET == old_ep_status_)) {
|
||||
speech_end_time_us_ = -1;
|
||||
waiting_for_speech_possibly_complete_timeout_ = false;
|
||||
waiting_for_speech_complete_timeout_ = false;
|
||||
// Trigger SpeechInputDidStart event on first detection.
|
||||
if (false == speech_previously_detected_) {
|
||||
speech_previously_detected_ = true;
|
||||
speech_start_time_us_ = ep_time;
|
||||
}
|
||||
}
|
||||
if ((EP_PRE_SPEECH == ep_status) &&
|
||||
(EP_POSSIBLE_OFFSET == old_ep_status_)) {
|
||||
speech_end_time_us_ = ep_time;
|
||||
waiting_for_speech_possibly_complete_timeout_ = true;
|
||||
waiting_for_speech_complete_timeout_ = true;
|
||||
}
|
||||
if (ep_time > speech_input_minimum_length_us_) {
|
||||
// Speech possibly complete timeout.
|
||||
if ((waiting_for_speech_possibly_complete_timeout_) &&
|
||||
(ep_time - speech_end_time_us_ >
|
||||
speech_input_possibly_complete_silence_length_us_)) {
|
||||
waiting_for_speech_possibly_complete_timeout_ = false;
|
||||
}
|
||||
if (waiting_for_speech_complete_timeout_) {
|
||||
// The length of the silence timeout period can be held constant, or it
|
||||
// can be changed after a fixed amount of time from the beginning of
|
||||
// speech.
|
||||
bool has_stepped_silence =
|
||||
(long_speech_length_us_ > 0) &&
|
||||
(long_speech_input_complete_silence_length_us_ > 0);
|
||||
int64_t requested_silence_length;
|
||||
if (has_stepped_silence &&
|
||||
(ep_time - speech_start_time_us_) > long_speech_length_us_) {
|
||||
requested_silence_length =
|
||||
long_speech_input_complete_silence_length_us_;
|
||||
} else {
|
||||
requested_silence_length =
|
||||
speech_input_complete_silence_length_us_;
|
||||
}
|
||||
|
||||
// Speech complete timeout.
|
||||
if ((ep_time - speech_end_time_us_) > requested_silence_length) {
|
||||
waiting_for_speech_complete_timeout_ = false;
|
||||
speech_input_complete_ = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
old_ep_status_ = ep_status;
|
||||
}
|
||||
return ep_status;
|
||||
}
|
||||
|
||||
} // namespace mozilla
|
||||
180
dom/media/webspeech/recognition/endpointer.h
Normal file
180
dom/media/webspeech/recognition/endpointer.h
Normal file
|
|
@ -0,0 +1,180 @@
|
|||
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without
|
||||
// modification, are permitted provided that the following conditions are
|
||||
// met:
|
||||
//
|
||||
// * Redistributions of source code must retain the above copyright
|
||||
// notice, this list of conditions and the following disclaimer.
|
||||
// * Redistributions in binary form must reproduce the above
|
||||
// copyright notice, this list of conditions and the following disclaimer
|
||||
// in the documentation and/or other materials provided with the
|
||||
// distribution.
|
||||
// * Neither the name of Google Inc. nor the names of its
|
||||
// contributors may be used to endorse or promote products derived from
|
||||
// this software without specific prior written permission.
|
||||
//
|
||||
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
|
||||
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
|
||||
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
||||
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
|
||||
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
||||
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
||||
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
#ifndef CONTENT_BROWSER_SPEECH_ENDPOINTER_ENDPOINTER_H_
|
||||
#define CONTENT_BROWSER_SPEECH_ENDPOINTER_ENDPOINTER_H_
|
||||
|
||||
#include "energy_endpointer.h"
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
struct AudioChunk;
|
||||
|
||||
// A simple interface to the underlying energy-endpointer implementation, this
|
||||
// class lets callers provide audio as being recorded and let them poll to find
|
||||
// when the user has stopped speaking.
|
||||
//
|
||||
// There are two events that may trigger the end of speech:
|
||||
//
|
||||
// speechInputPossiblyComplete event:
|
||||
//
|
||||
// Signals that silence/noise has been detected for a *short* amount of
|
||||
// time after some speech has been detected. It can be used for low latency
|
||||
// UI feedback. To disable it, set it to a large amount.
|
||||
//
|
||||
// speechInputComplete event:
|
||||
//
|
||||
// This event is intended to signal end of input and to stop recording.
|
||||
// The amount of time to wait after speech is set by
|
||||
// speech_input_complete_silence_length_ and optionally two other
|
||||
// parameters (see below).
|
||||
// This time can be held constant, or can change as more speech is detected.
|
||||
// In the latter case, the time changes after a set amount of time from the
|
||||
// *beginning* of speech. This is motivated by the expectation that there
|
||||
// will be two distinct types of inputs: short search queries and longer
|
||||
// dictation style input.
|
||||
//
|
||||
// Three parameters are used to define the piecewise constant timeout function.
|
||||
// The timeout length is speech_input_complete_silence_length until
|
||||
// long_speech_length, when it changes to
|
||||
// long_speech_input_complete_silence_length.
|
||||
class Endpointer {
|
||||
public:
|
||||
explicit Endpointer(int sample_rate);
|
||||
|
||||
// Start the endpointer. This should be called at the beginning of a session.
|
||||
void StartSession();
|
||||
|
||||
// Stop the endpointer.
|
||||
void EndSession();
|
||||
|
||||
// Start environment estimation. Audio will be used for environment estimation
|
||||
// i.e. noise level estimation.
|
||||
void SetEnvironmentEstimationMode();
|
||||
|
||||
// Start user input. This should be called when the user indicates start of
|
||||
// input, e.g. by pressing a button.
|
||||
void SetUserInputMode();
|
||||
|
||||
// Process a segment of audio, which may be more than one frame.
|
||||
// The status of the last frame will be returned.
|
||||
EpStatus ProcessAudio(const AudioChunk& raw_audio, float* rms_out);
|
||||
|
||||
// Get the status of the endpointer.
|
||||
EpStatus Status(int64_t *time_us);
|
||||
|
||||
// Get the expected frame size for audio chunks. Audio chunks are expected
|
||||
// to contain a number of samples that is a multiple of this number, and extra
|
||||
// samples will be dropped.
|
||||
int32_t FrameSize() const {
|
||||
return frame_size_;
|
||||
}
|
||||
|
||||
// Returns true if the endpointer detected reasonable audio levels above
|
||||
// background noise which could be user speech, false if not.
|
||||
bool DidStartReceivingSpeech() const {
|
||||
return speech_previously_detected_;
|
||||
}
|
||||
|
||||
bool IsEstimatingEnvironment() const {
|
||||
return energy_endpointer_.estimating_environment();
|
||||
}
|
||||
|
||||
void set_speech_input_complete_silence_length(int64_t time_us) {
|
||||
speech_input_complete_silence_length_us_ = time_us;
|
||||
}
|
||||
|
||||
void set_long_speech_input_complete_silence_length(int64_t time_us) {
|
||||
long_speech_input_complete_silence_length_us_ = time_us;
|
||||
}
|
||||
|
||||
void set_speech_input_possibly_complete_silence_length(int64_t time_us) {
|
||||
speech_input_possibly_complete_silence_length_us_ = time_us;
|
||||
}
|
||||
|
||||
void set_long_speech_length(int64_t time_us) {
|
||||
long_speech_length_us_ = time_us;
|
||||
}
|
||||
|
||||
bool speech_input_complete() const {
|
||||
return speech_input_complete_;
|
||||
}
|
||||
|
||||
// RMS background noise level in dB.
|
||||
float NoiseLevelDb() const { return energy_endpointer_.GetNoiseLevelDb(); }
|
||||
|
||||
private:
|
||||
// Reset internal states. Helper method common to initial input utterance
|
||||
// and following input utternaces.
|
||||
void Reset();
|
||||
|
||||
// Minimum allowable length of speech input.
|
||||
int64_t speech_input_minimum_length_us_;
|
||||
|
||||
// The speechInputPossiblyComplete event signals that silence/noise has been
|
||||
// detected for a *short* amount of time after some speech has been detected.
|
||||
// This proporty specifies the time period.
|
||||
int64_t speech_input_possibly_complete_silence_length_us_;
|
||||
|
||||
// The speechInputComplete event signals that silence/noise has been
|
||||
// detected for a *long* amount of time after some speech has been detected.
|
||||
// This property specifies the time period.
|
||||
int64_t speech_input_complete_silence_length_us_;
|
||||
|
||||
// Same as above, this specifies the required silence period after speech
|
||||
// detection. This period is used instead of
|
||||
// speech_input_complete_silence_length_ when the utterance is longer than
|
||||
// long_speech_length_. This parameter is optional.
|
||||
int64_t long_speech_input_complete_silence_length_us_;
|
||||
|
||||
// The period of time after which the endpointer should consider
|
||||
// long_speech_input_complete_silence_length_ as a valid silence period
|
||||
// instead of speech_input_complete_silence_length_. This parameter is
|
||||
// optional.
|
||||
int64_t long_speech_length_us_;
|
||||
|
||||
// First speech onset time, used in determination of speech complete timeout.
|
||||
int64_t speech_start_time_us_;
|
||||
|
||||
// Most recent end time, used in determination of speech complete timeout.
|
||||
int64_t speech_end_time_us_;
|
||||
|
||||
int64_t audio_frame_time_us_;
|
||||
EpStatus old_ep_status_;
|
||||
bool waiting_for_speech_possibly_complete_timeout_;
|
||||
bool waiting_for_speech_complete_timeout_;
|
||||
bool speech_previously_detected_;
|
||||
bool speech_input_complete_;
|
||||
EnergyEndpointer energy_endpointer_;
|
||||
int sample_rate_;
|
||||
int32_t frame_size_;
|
||||
};
|
||||
|
||||
} // namespace mozilla
|
||||
|
||||
#endif // CONTENT_BROWSER_SPEECH_ENDPOINTER_ENDPOINTER_H_
|
||||
393
dom/media/webspeech/recognition/energy_endpointer.cc
Normal file
393
dom/media/webspeech/recognition/energy_endpointer.cc
Normal file
|
|
@ -0,0 +1,393 @@
|
|||
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without
|
||||
// modification, are permitted provided that the following conditions are
|
||||
// met:
|
||||
//
|
||||
// * Redistributions of source code must retain the above copyright
|
||||
// notice, this list of conditions and the following disclaimer.
|
||||
// * Redistributions in binary form must reproduce the above
|
||||
// copyright notice, this list of conditions and the following disclaimer
|
||||
// in the documentation and/or other materials provided with the
|
||||
// distribution.
|
||||
// * Neither the name of Google Inc. nor the names of its
|
||||
// contributors may be used to endorse or promote products derived from
|
||||
// this software without specific prior written permission.
|
||||
//
|
||||
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
|
||||
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
|
||||
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
||||
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
|
||||
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
||||
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
||||
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
#include "energy_endpointer.h"
|
||||
|
||||
#include <math.h>
|
||||
|
||||
namespace {
|
||||
|
||||
// Returns the RMS (quadratic mean) of the input signal.
|
||||
float RMS(const int16_t* samples, int num_samples) {
|
||||
int64_t ssq_int64_t = 0;
|
||||
int64_t sum_int64_t = 0;
|
||||
for (int i = 0; i < num_samples; ++i) {
|
||||
sum_int64_t += samples[i];
|
||||
ssq_int64_t += samples[i] * samples[i];
|
||||
}
|
||||
// now convert to floats.
|
||||
double sum = static_cast<double>(sum_int64_t);
|
||||
sum /= num_samples;
|
||||
double ssq = static_cast<double>(ssq_int64_t);
|
||||
return static_cast<float>(sqrt((ssq / num_samples) - (sum * sum)));
|
||||
}
|
||||
|
||||
int64_t Secs2Usecs(float seconds) {
|
||||
return static_cast<int64_t>(0.5 + (1.0e6 * seconds));
|
||||
}
|
||||
|
||||
float GetDecibel(float value) {
|
||||
if (value > 1.0e-100)
|
||||
return 20 * log10(value);
|
||||
return -2000.0;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
// Stores threshold-crossing histories for making decisions about the speech
|
||||
// state.
|
||||
class EnergyEndpointer::HistoryRing {
|
||||
public:
|
||||
HistoryRing() : insertion_index_(0) {}
|
||||
|
||||
// Resets the ring to |size| elements each with state |initial_state|
|
||||
void SetRing(int size, bool initial_state);
|
||||
|
||||
// Inserts a new entry into the ring and drops the oldest entry.
|
||||
void Insert(int64_t time_us, bool decision);
|
||||
|
||||
// Returns the time in microseconds of the most recently added entry.
|
||||
int64_t EndTime() const;
|
||||
|
||||
// Returns the sum of all intervals during which 'decision' is true within
|
||||
// the time in seconds specified by 'duration'. The returned interval is
|
||||
// in seconds.
|
||||
float RingSum(float duration_sec);
|
||||
|
||||
private:
|
||||
struct DecisionPoint {
|
||||
int64_t time_us;
|
||||
bool decision;
|
||||
};
|
||||
|
||||
std::vector<DecisionPoint> decision_points_;
|
||||
int insertion_index_; // Index at which the next item gets added/inserted.
|
||||
|
||||
HistoryRing(const HistoryRing&);
|
||||
void operator=(const HistoryRing&);
|
||||
};
|
||||
|
||||
void EnergyEndpointer::HistoryRing::SetRing(int size, bool initial_state) {
|
||||
insertion_index_ = 0;
|
||||
decision_points_.clear();
|
||||
DecisionPoint init = { -1, initial_state };
|
||||
decision_points_.resize(size, init);
|
||||
}
|
||||
|
||||
void EnergyEndpointer::HistoryRing::Insert(int64_t time_us, bool decision) {
|
||||
decision_points_[insertion_index_].time_us = time_us;
|
||||
decision_points_[insertion_index_].decision = decision;
|
||||
insertion_index_ = (insertion_index_ + 1) % decision_points_.size();
|
||||
}
|
||||
|
||||
int64_t EnergyEndpointer::HistoryRing::EndTime() const {
|
||||
int ind = insertion_index_ - 1;
|
||||
if (ind < 0)
|
||||
ind = decision_points_.size() - 1;
|
||||
return decision_points_[ind].time_us;
|
||||
}
|
||||
|
||||
float EnergyEndpointer::HistoryRing::RingSum(float duration_sec) {
|
||||
if (!decision_points_.size())
|
||||
return 0.0;
|
||||
|
||||
int64_t sum_us = 0;
|
||||
int ind = insertion_index_ - 1;
|
||||
if (ind < 0)
|
||||
ind = decision_points_.size() - 1;
|
||||
int64_t end_us = decision_points_[ind].time_us;
|
||||
bool is_on = decision_points_[ind].decision;
|
||||
int64_t start_us = end_us - static_cast<int64_t>(0.5 + (1.0e6 * duration_sec));
|
||||
if (start_us < 0)
|
||||
start_us = 0;
|
||||
size_t n_summed = 1; // n points ==> (n-1) intervals
|
||||
while ((decision_points_[ind].time_us > start_us) &&
|
||||
(n_summed < decision_points_.size())) {
|
||||
--ind;
|
||||
if (ind < 0)
|
||||
ind = decision_points_.size() - 1;
|
||||
if (is_on)
|
||||
sum_us += end_us - decision_points_[ind].time_us;
|
||||
is_on = decision_points_[ind].decision;
|
||||
end_us = decision_points_[ind].time_us;
|
||||
n_summed++;
|
||||
}
|
||||
|
||||
return 1.0e-6f * sum_us; // Returns total time that was super threshold.
|
||||
}
|
||||
|
||||
EnergyEndpointer::EnergyEndpointer()
|
||||
: status_(EP_PRE_SPEECH),
|
||||
offset_confirm_dur_sec_(0),
|
||||
endpointer_time_us_(0),
|
||||
fast_update_frames_(0),
|
||||
frame_counter_(0),
|
||||
max_window_dur_(4.0),
|
||||
sample_rate_(0),
|
||||
history_(new HistoryRing()),
|
||||
decision_threshold_(0),
|
||||
estimating_environment_(false),
|
||||
noise_level_(0),
|
||||
rms_adapt_(0),
|
||||
start_lag_(0),
|
||||
end_lag_(0),
|
||||
user_input_start_time_us_(0) {
|
||||
}
|
||||
|
||||
EnergyEndpointer::~EnergyEndpointer() {
|
||||
}
|
||||
|
||||
int EnergyEndpointer::TimeToFrame(float time) const {
|
||||
return static_cast<int32_t>(0.5 + (time / params_.frame_period()));
|
||||
}
|
||||
|
||||
void EnergyEndpointer::Restart(bool reset_threshold) {
|
||||
status_ = EP_PRE_SPEECH;
|
||||
user_input_start_time_us_ = 0;
|
||||
|
||||
if (reset_threshold) {
|
||||
decision_threshold_ = params_.decision_threshold();
|
||||
rms_adapt_ = decision_threshold_;
|
||||
noise_level_ = params_.decision_threshold() / 2.0f;
|
||||
frame_counter_ = 0; // Used for rapid initial update of levels.
|
||||
}
|
||||
|
||||
// Set up the memories to hold the history windows.
|
||||
history_->SetRing(TimeToFrame(max_window_dur_), false);
|
||||
|
||||
// Flag that indicates that current input should be used for
|
||||
// estimating the environment. The user has not yet started input
|
||||
// by e.g. pressed the push-to-talk button. By default, this is
|
||||
// false for backward compatibility.
|
||||
estimating_environment_ = false;
|
||||
}
|
||||
|
||||
void EnergyEndpointer::Init(const EnergyEndpointerParams& params) {
|
||||
params_ = params;
|
||||
|
||||
// Find the longest history interval to be used, and make the ring
|
||||
// large enough to accommodate that number of frames. NOTE: This
|
||||
// depends upon ep_frame_period being set correctly in the factory
|
||||
// that did this instantiation.
|
||||
max_window_dur_ = params_.onset_window();
|
||||
if (params_.speech_on_window() > max_window_dur_)
|
||||
max_window_dur_ = params_.speech_on_window();
|
||||
if (params_.offset_window() > max_window_dur_)
|
||||
max_window_dur_ = params_.offset_window();
|
||||
Restart(true);
|
||||
|
||||
offset_confirm_dur_sec_ = params_.offset_window() -
|
||||
params_.offset_confirm_dur();
|
||||
if (offset_confirm_dur_sec_ < 0.0)
|
||||
offset_confirm_dur_sec_ = 0.0;
|
||||
|
||||
user_input_start_time_us_ = 0;
|
||||
|
||||
// Flag that indicates that current input should be used for
|
||||
// estimating the environment. The user has not yet started input
|
||||
// by e.g. pressed the push-to-talk button. By default, this is
|
||||
// false for backward compatibility.
|
||||
estimating_environment_ = false;
|
||||
// The initial value of the noise and speech levels is inconsequential.
|
||||
// The level of the first frame will overwrite these values.
|
||||
noise_level_ = params_.decision_threshold() / 2.0f;
|
||||
fast_update_frames_ =
|
||||
static_cast<int64_t>(params_.fast_update_dur() / params_.frame_period());
|
||||
|
||||
frame_counter_ = 0; // Used for rapid initial update of levels.
|
||||
|
||||
sample_rate_ = params_.sample_rate();
|
||||
start_lag_ = static_cast<int>(sample_rate_ /
|
||||
params_.max_fundamental_frequency());
|
||||
end_lag_ = static_cast<int>(sample_rate_ /
|
||||
params_.min_fundamental_frequency());
|
||||
}
|
||||
|
||||
void EnergyEndpointer::StartSession() {
|
||||
Restart(true);
|
||||
}
|
||||
|
||||
void EnergyEndpointer::EndSession() {
|
||||
status_ = EP_POST_SPEECH;
|
||||
}
|
||||
|
||||
void EnergyEndpointer::SetEnvironmentEstimationMode() {
|
||||
Restart(true);
|
||||
estimating_environment_ = true;
|
||||
}
|
||||
|
||||
void EnergyEndpointer::SetUserInputMode() {
|
||||
estimating_environment_ = false;
|
||||
user_input_start_time_us_ = endpointer_time_us_;
|
||||
}
|
||||
|
||||
void EnergyEndpointer::ProcessAudioFrame(int64_t time_us,
|
||||
const int16_t* samples,
|
||||
int num_samples,
|
||||
float* rms_out) {
|
||||
endpointer_time_us_ = time_us;
|
||||
float rms = RMS(samples, num_samples);
|
||||
|
||||
// Check that this is user input audio vs. pre-input adaptation audio.
|
||||
// Input audio starts when the user indicates start of input, by e.g.
|
||||
// pressing push-to-talk. Audio recieved prior to that is used to update
|
||||
// noise and speech level estimates.
|
||||
if (!estimating_environment_) {
|
||||
bool decision = false;
|
||||
if ((endpointer_time_us_ - user_input_start_time_us_) <
|
||||
Secs2Usecs(params_.contamination_rejection_period())) {
|
||||
decision = false;
|
||||
//PR_LOG(GetSpeechRecognitionLog(), PR_LOG_DEBUG, ("decision: forced to false, time: %d", endpointer_time_us_));
|
||||
} else {
|
||||
decision = (rms > decision_threshold_);
|
||||
}
|
||||
|
||||
history_->Insert(endpointer_time_us_, decision);
|
||||
|
||||
switch (status_) {
|
||||
case EP_PRE_SPEECH:
|
||||
if (history_->RingSum(params_.onset_window()) >
|
||||
params_.onset_detect_dur()) {
|
||||
status_ = EP_POSSIBLE_ONSET;
|
||||
}
|
||||
break;
|
||||
|
||||
case EP_POSSIBLE_ONSET: {
|
||||
float tsum = history_->RingSum(params_.onset_window());
|
||||
if (tsum > params_.onset_confirm_dur()) {
|
||||
status_ = EP_SPEECH_PRESENT;
|
||||
} else { // If signal is not maintained, drop back to pre-speech.
|
||||
if (tsum <= params_.onset_detect_dur())
|
||||
status_ = EP_PRE_SPEECH;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case EP_SPEECH_PRESENT: {
|
||||
// To induce hysteresis in the state residency, we allow a
|
||||
// smaller residency time in the on_ring, than was required to
|
||||
// enter the SPEECH_PERSENT state.
|
||||
float on_time = history_->RingSum(params_.speech_on_window());
|
||||
if (on_time < params_.on_maintain_dur())
|
||||
status_ = EP_POSSIBLE_OFFSET;
|
||||
break;
|
||||
}
|
||||
|
||||
case EP_POSSIBLE_OFFSET:
|
||||
if (history_->RingSum(params_.offset_window()) <=
|
||||
offset_confirm_dur_sec_) {
|
||||
// Note that this offset time may be beyond the end
|
||||
// of the input buffer in a real-time system. It will be up
|
||||
// to the RecognizerSession to decide what to do.
|
||||
status_ = EP_PRE_SPEECH; // Automatically reset for next utterance.
|
||||
} else { // If speech picks up again we allow return to SPEECH_PRESENT.
|
||||
if (history_->RingSum(params_.speech_on_window()) >=
|
||||
params_.on_maintain_dur())
|
||||
status_ = EP_SPEECH_PRESENT;
|
||||
}
|
||||
break;
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
// If this is a quiet, non-speech region, slowly adapt the detection
|
||||
// threshold to be about 6dB above the average RMS.
|
||||
if ((!decision) && (status_ == EP_PRE_SPEECH)) {
|
||||
decision_threshold_ = (0.98f * decision_threshold_) + (0.02f * 2 * rms);
|
||||
rms_adapt_ = decision_threshold_;
|
||||
} else {
|
||||
// If this is in a speech region, adapt the decision threshold to
|
||||
// be about 10dB below the average RMS. If the noise level is high,
|
||||
// the threshold is pushed up.
|
||||
// Adaptation up to a higher level is 5 times faster than decay to
|
||||
// a lower level.
|
||||
if ((status_ == EP_SPEECH_PRESENT) && decision) {
|
||||
if (rms_adapt_ > rms) {
|
||||
rms_adapt_ = (0.99f * rms_adapt_) + (0.01f * rms);
|
||||
} else {
|
||||
rms_adapt_ = (0.95f * rms_adapt_) + (0.05f * rms);
|
||||
}
|
||||
float target_threshold = 0.3f * rms_adapt_ + noise_level_;
|
||||
decision_threshold_ = (.90f * decision_threshold_) +
|
||||
(0.10f * target_threshold);
|
||||
}
|
||||
}
|
||||
|
||||
// Set a floor
|
||||
if (decision_threshold_ < params_.min_decision_threshold())
|
||||
decision_threshold_ = params_.min_decision_threshold();
|
||||
}
|
||||
|
||||
// Update speech and noise levels.
|
||||
UpdateLevels(rms);
|
||||
++frame_counter_;
|
||||
|
||||
if (rms_out)
|
||||
*rms_out = GetDecibel(rms);
|
||||
}
|
||||
|
||||
float EnergyEndpointer::GetNoiseLevelDb() const {
|
||||
return GetDecibel(noise_level_);
|
||||
}
|
||||
|
||||
void EnergyEndpointer::UpdateLevels(float rms) {
|
||||
// Update quickly initially. We assume this is noise and that
|
||||
// speech is 6dB above the noise.
|
||||
if (frame_counter_ < fast_update_frames_) {
|
||||
// Alpha increases from 0 to (k-1)/k where k is the number of time
|
||||
// steps in the initial adaptation period.
|
||||
float alpha = static_cast<float>(frame_counter_) /
|
||||
static_cast<float>(fast_update_frames_);
|
||||
noise_level_ = (alpha * noise_level_) + ((1 - alpha) * rms);
|
||||
//PR_LOG(GetSpeechRecognitionLog(), PR_LOG_DEBUG, ("FAST UPDATE, frame_counter_ %d, fast_update_frames_ %d", frame_counter_, fast_update_frames_));
|
||||
} else {
|
||||
// Update Noise level. The noise level adapts quickly downward, but
|
||||
// slowly upward. The noise_level_ parameter is not currently used
|
||||
// for threshold adaptation. It is used for UI feedback.
|
||||
if (noise_level_ < rms)
|
||||
noise_level_ = (0.999f * noise_level_) + (0.001f * rms);
|
||||
else
|
||||
noise_level_ = (0.95f * noise_level_) + (0.05f * rms);
|
||||
}
|
||||
if (estimating_environment_ || (frame_counter_ < fast_update_frames_)) {
|
||||
decision_threshold_ = noise_level_ * 2; // 6dB above noise level.
|
||||
// Set a floor
|
||||
if (decision_threshold_ < params_.min_decision_threshold())
|
||||
decision_threshold_ = params_.min_decision_threshold();
|
||||
}
|
||||
}
|
||||
|
||||
EpStatus EnergyEndpointer::Status(int64_t* status_time) const {
|
||||
*status_time = history_->EndTime();
|
||||
return status_;
|
||||
}
|
||||
|
||||
} // namespace mozilla
|
||||
180
dom/media/webspeech/recognition/energy_endpointer.h
Normal file
180
dom/media/webspeech/recognition/energy_endpointer.h
Normal file
|
|
@ -0,0 +1,180 @@
|
|||
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without
|
||||
// modification, are permitted provided that the following conditions are
|
||||
// met:
|
||||
//
|
||||
// * Redistributions of source code must retain the above copyright
|
||||
// notice, this list of conditions and the following disclaimer.
|
||||
// * Redistributions in binary form must reproduce the above
|
||||
// copyright notice, this list of conditions and the following disclaimer
|
||||
// in the documentation and/or other materials provided with the
|
||||
// distribution.
|
||||
// * Neither the name of Google Inc. nor the names of its
|
||||
// contributors may be used to endorse or promote products derived from
|
||||
// this software without specific prior written permission.
|
||||
//
|
||||
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
|
||||
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
|
||||
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
||||
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
|
||||
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
||||
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
||||
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
// The EnergyEndpointer class finds likely speech onset and offset points.
|
||||
//
|
||||
// The implementation described here is about the simplest possible.
|
||||
// It is based on timings of threshold crossings for overall signal
|
||||
// RMS. It is suitable for light weight applications.
|
||||
//
|
||||
// As written, the basic idea is that one specifies intervals that
|
||||
// must be occupied by super- and sub-threshold energy levels, and
|
||||
// defers decisions re onset and offset times until these
|
||||
// specifications have been met. Three basic intervals are tested: an
|
||||
// onset window, a speech-on window, and an offset window. We require
|
||||
// super-threshold to exceed some mimimum total durations in the onset
|
||||
// and speech-on windows before declaring the speech onset time, and
|
||||
// we specify a required sub-threshold residency in the offset window
|
||||
// before declaring speech offset. As the various residency requirements are
|
||||
// met, the EnergyEndpointer instance assumes various states, and can return the
|
||||
// ID of these states to the client (see EpStatus below).
|
||||
//
|
||||
// The levels of the speech and background noise are continuously updated. It is
|
||||
// important that the background noise level be estimated initially for
|
||||
// robustness in noisy conditions. The first frames are assumed to be background
|
||||
// noise and a fast update rate is used for the noise level. The duration for
|
||||
// fast update is controlled by the fast_update_dur_ paramter.
|
||||
//
|
||||
// If used in noisy conditions, the endpointer should be started and run in the
|
||||
// EnvironmentEstimation mode, for at least 200ms, before switching to
|
||||
// UserInputMode.
|
||||
// Audio feedback contamination can appear in the input audio, if not cut
|
||||
// out or handled by echo cancellation. Audio feedback can trigger a false
|
||||
// accept. The false accepts can be ignored by setting
|
||||
// ep_contamination_rejection_period.
|
||||
|
||||
#ifndef CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_H_
|
||||
#define CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_H_
|
||||
|
||||
#include <vector>
|
||||
|
||||
#include "nsAutoPtr.h"
|
||||
|
||||
#include "energy_endpointer_params.h"
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
// Endpointer status codes
|
||||
enum EpStatus {
|
||||
EP_PRE_SPEECH = 10,
|
||||
EP_POSSIBLE_ONSET,
|
||||
EP_SPEECH_PRESENT,
|
||||
EP_POSSIBLE_OFFSET,
|
||||
EP_POST_SPEECH,
|
||||
};
|
||||
|
||||
class EnergyEndpointer {
|
||||
public:
|
||||
// The default construction MUST be followed by Init(), before any
|
||||
// other use can be made of the instance.
|
||||
EnergyEndpointer();
|
||||
virtual ~EnergyEndpointer();
|
||||
|
||||
void Init(const EnergyEndpointerParams& params);
|
||||
|
||||
// Start the endpointer. This should be called at the beginning of a session.
|
||||
void StartSession();
|
||||
|
||||
// Stop the endpointer.
|
||||
void EndSession();
|
||||
|
||||
// Start environment estimation. Audio will be used for environment estimation
|
||||
// i.e. noise level estimation.
|
||||
void SetEnvironmentEstimationMode();
|
||||
|
||||
// Start user input. This should be called when the user indicates start of
|
||||
// input, e.g. by pressing a button.
|
||||
void SetUserInputMode();
|
||||
|
||||
// Computes the next input frame and modifies EnergyEndpointer status as
|
||||
// appropriate based on the computation.
|
||||
void ProcessAudioFrame(int64_t time_us,
|
||||
const int16_t* samples, int num_samples,
|
||||
float* rms_out);
|
||||
|
||||
// Returns the current state of the EnergyEndpointer and the time
|
||||
// corresponding to the most recently computed frame.
|
||||
EpStatus Status(int64_t* status_time_us) const;
|
||||
|
||||
bool estimating_environment() const {
|
||||
return estimating_environment_;
|
||||
}
|
||||
|
||||
// Returns estimated noise level in dB.
|
||||
float GetNoiseLevelDb() const;
|
||||
|
||||
private:
|
||||
class HistoryRing;
|
||||
|
||||
// Resets the endpointer internal state. If reset_threshold is true, the
|
||||
// state will be reset completely, including adaptive thresholds and the
|
||||
// removal of all history information.
|
||||
void Restart(bool reset_threshold);
|
||||
|
||||
// Update internal speech and noise levels.
|
||||
void UpdateLevels(float rms);
|
||||
|
||||
// Returns the number of frames (or frame number) corresponding to
|
||||
// the 'time' (in seconds).
|
||||
int TimeToFrame(float time) const;
|
||||
|
||||
EpStatus status_; // The current state of this instance.
|
||||
float offset_confirm_dur_sec_; // max on time allowed to confirm POST_SPEECH
|
||||
int64_t endpointer_time_us_; // Time of the most recently received audio frame.
|
||||
int64_t fast_update_frames_; // Number of frames for initial level adaptation.
|
||||
int64_t frame_counter_; // Number of frames seen. Used for initial adaptation.
|
||||
float max_window_dur_; // Largest search window size (seconds)
|
||||
float sample_rate_; // Sampling rate.
|
||||
|
||||
// Ring buffers to hold the speech activity history.
|
||||
nsAutoPtr<HistoryRing> history_;
|
||||
|
||||
// Configuration parameters.
|
||||
EnergyEndpointerParams params_;
|
||||
|
||||
// RMS which must be exceeded to conclude frame is speech.
|
||||
float decision_threshold_;
|
||||
|
||||
// Flag to indicate that audio should be used to estimate environment, prior
|
||||
// to receiving user input.
|
||||
bool estimating_environment_;
|
||||
|
||||
// Estimate of the background noise level. Used externally for UI feedback.
|
||||
float noise_level_;
|
||||
|
||||
// An adaptive threshold used to update decision_threshold_ when appropriate.
|
||||
float rms_adapt_;
|
||||
|
||||
// Start lag corresponds to the highest fundamental frequency.
|
||||
int start_lag_;
|
||||
|
||||
// End lag corresponds to the lowest fundamental frequency.
|
||||
int end_lag_;
|
||||
|
||||
// Time when mode switched from environment estimation to user input. This
|
||||
// is used to time forced rejection of audio feedback contamination.
|
||||
int64_t user_input_start_time_us_;
|
||||
|
||||
// prevent copy constructor and assignment
|
||||
EnergyEndpointer(const EnergyEndpointer&);
|
||||
void operator=(const EnergyEndpointer&);
|
||||
};
|
||||
|
||||
} // namespace mozilla
|
||||
|
||||
#endif // CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_H_
|
||||
77
dom/media/webspeech/recognition/energy_endpointer_params.cc
Normal file
77
dom/media/webspeech/recognition/energy_endpointer_params.cc
Normal file
|
|
@ -0,0 +1,77 @@
|
|||
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without
|
||||
// modification, are permitted provided that the following conditions are
|
||||
// met:
|
||||
//
|
||||
// * Redistributions of source code must retain the above copyright
|
||||
// notice, this list of conditions and the following disclaimer.
|
||||
// * Redistributions in binary form must reproduce the above
|
||||
// copyright notice, this list of conditions and the following disclaimer
|
||||
// in the documentation and/or other materials provided with the
|
||||
// distribution.
|
||||
// * Neither the name of Google Inc. nor the names of its
|
||||
// contributors may be used to endorse or promote products derived from
|
||||
// this software without specific prior written permission.
|
||||
//
|
||||
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
|
||||
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
|
||||
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
||||
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
|
||||
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
||||
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
||||
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
#include "energy_endpointer_params.h"
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
EnergyEndpointerParams::EnergyEndpointerParams() {
|
||||
SetDefaults();
|
||||
}
|
||||
|
||||
void EnergyEndpointerParams::SetDefaults() {
|
||||
frame_period_ = 0.01f;
|
||||
frame_duration_ = 0.01f;
|
||||
endpoint_margin_ = 0.2f;
|
||||
onset_window_ = 0.15f;
|
||||
speech_on_window_ = 0.4f;
|
||||
offset_window_ = 0.15f;
|
||||
onset_detect_dur_ = 0.09f;
|
||||
onset_confirm_dur_ = 0.075f;
|
||||
on_maintain_dur_ = 0.10f;
|
||||
offset_confirm_dur_ = 0.12f;
|
||||
decision_threshold_ = 150.0f;
|
||||
min_decision_threshold_ = 50.0f;
|
||||
fast_update_dur_ = 0.2f;
|
||||
sample_rate_ = 8000.0f;
|
||||
min_fundamental_frequency_ = 57.143f;
|
||||
max_fundamental_frequency_ = 400.0f;
|
||||
contamination_rejection_period_ = 0.25f;
|
||||
}
|
||||
|
||||
void EnergyEndpointerParams::operator=(const EnergyEndpointerParams& source) {
|
||||
frame_period_ = source.frame_period();
|
||||
frame_duration_ = source.frame_duration();
|
||||
endpoint_margin_ = source.endpoint_margin();
|
||||
onset_window_ = source.onset_window();
|
||||
speech_on_window_ = source.speech_on_window();
|
||||
offset_window_ = source.offset_window();
|
||||
onset_detect_dur_ = source.onset_detect_dur();
|
||||
onset_confirm_dur_ = source.onset_confirm_dur();
|
||||
on_maintain_dur_ = source.on_maintain_dur();
|
||||
offset_confirm_dur_ = source.offset_confirm_dur();
|
||||
decision_threshold_ = source.decision_threshold();
|
||||
min_decision_threshold_ = source.min_decision_threshold();
|
||||
fast_update_dur_ = source.fast_update_dur();
|
||||
sample_rate_ = source.sample_rate();
|
||||
min_fundamental_frequency_ = source.min_fundamental_frequency();
|
||||
max_fundamental_frequency_ = source.max_fundamental_frequency();
|
||||
contamination_rejection_period_ = source.contamination_rejection_period();
|
||||
}
|
||||
|
||||
} // namespace mozilla
|
||||
159
dom/media/webspeech/recognition/energy_endpointer_params.h
Normal file
159
dom/media/webspeech/recognition/energy_endpointer_params.h
Normal file
|
|
@ -0,0 +1,159 @@
|
|||
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without
|
||||
// modification, are permitted provided that the following conditions are
|
||||
// met:
|
||||
//
|
||||
// * Redistributions of source code must retain the above copyright
|
||||
// notice, this list of conditions and the following disclaimer.
|
||||
// * Redistributions in binary form must reproduce the above
|
||||
// copyright notice, this list of conditions and the following disclaimer
|
||||
// in the documentation and/or other materials provided with the
|
||||
// distribution.
|
||||
// * Neither the name of Google Inc. nor the names of its
|
||||
// contributors may be used to endorse or promote products derived from
|
||||
// this software without specific prior written permission.
|
||||
//
|
||||
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
|
||||
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
|
||||
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
||||
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
|
||||
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
||||
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
||||
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
#ifndef CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_PARAMS_H_
|
||||
#define CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_PARAMS_H_
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
// Input parameters for the EnergyEndpointer class.
|
||||
class EnergyEndpointerParams {
|
||||
public:
|
||||
EnergyEndpointerParams();
|
||||
|
||||
void SetDefaults();
|
||||
|
||||
void operator=(const EnergyEndpointerParams& source);
|
||||
|
||||
// Accessors and mutators
|
||||
float frame_period() const { return frame_period_; }
|
||||
void set_frame_period(float frame_period) {
|
||||
frame_period_ = frame_period;
|
||||
}
|
||||
|
||||
float frame_duration() const { return frame_duration_; }
|
||||
void set_frame_duration(float frame_duration) {
|
||||
frame_duration_ = frame_duration;
|
||||
}
|
||||
|
||||
float endpoint_margin() const { return endpoint_margin_; }
|
||||
void set_endpoint_margin(float endpoint_margin) {
|
||||
endpoint_margin_ = endpoint_margin;
|
||||
}
|
||||
|
||||
float onset_window() const { return onset_window_; }
|
||||
void set_onset_window(float onset_window) { onset_window_ = onset_window; }
|
||||
|
||||
float speech_on_window() const { return speech_on_window_; }
|
||||
void set_speech_on_window(float speech_on_window) {
|
||||
speech_on_window_ = speech_on_window;
|
||||
}
|
||||
|
||||
float offset_window() const { return offset_window_; }
|
||||
void set_offset_window(float offset_window) {
|
||||
offset_window_ = offset_window;
|
||||
}
|
||||
|
||||
float onset_detect_dur() const { return onset_detect_dur_; }
|
||||
void set_onset_detect_dur(float onset_detect_dur) {
|
||||
onset_detect_dur_ = onset_detect_dur;
|
||||
}
|
||||
|
||||
float onset_confirm_dur() const { return onset_confirm_dur_; }
|
||||
void set_onset_confirm_dur(float onset_confirm_dur) {
|
||||
onset_confirm_dur_ = onset_confirm_dur;
|
||||
}
|
||||
|
||||
float on_maintain_dur() const { return on_maintain_dur_; }
|
||||
void set_on_maintain_dur(float on_maintain_dur) {
|
||||
on_maintain_dur_ = on_maintain_dur;
|
||||
}
|
||||
|
||||
float offset_confirm_dur() const { return offset_confirm_dur_; }
|
||||
void set_offset_confirm_dur(float offset_confirm_dur) {
|
||||
offset_confirm_dur_ = offset_confirm_dur;
|
||||
}
|
||||
|
||||
float decision_threshold() const { return decision_threshold_; }
|
||||
void set_decision_threshold(float decision_threshold) {
|
||||
decision_threshold_ = decision_threshold;
|
||||
}
|
||||
|
||||
float min_decision_threshold() const { return min_decision_threshold_; }
|
||||
void set_min_decision_threshold(float min_decision_threshold) {
|
||||
min_decision_threshold_ = min_decision_threshold;
|
||||
}
|
||||
|
||||
float fast_update_dur() const { return fast_update_dur_; }
|
||||
void set_fast_update_dur(float fast_update_dur) {
|
||||
fast_update_dur_ = fast_update_dur;
|
||||
}
|
||||
|
||||
float sample_rate() const { return sample_rate_; }
|
||||
void set_sample_rate(float sample_rate) { sample_rate_ = sample_rate; }
|
||||
|
||||
float min_fundamental_frequency() const { return min_fundamental_frequency_; }
|
||||
void set_min_fundamental_frequency(float min_fundamental_frequency) {
|
||||
min_fundamental_frequency_ = min_fundamental_frequency;
|
||||
}
|
||||
|
||||
float max_fundamental_frequency() const { return max_fundamental_frequency_; }
|
||||
void set_max_fundamental_frequency(float max_fundamental_frequency) {
|
||||
max_fundamental_frequency_ = max_fundamental_frequency;
|
||||
}
|
||||
|
||||
float contamination_rejection_period() const {
|
||||
return contamination_rejection_period_;
|
||||
}
|
||||
void set_contamination_rejection_period(
|
||||
float contamination_rejection_period) {
|
||||
contamination_rejection_period_ = contamination_rejection_period;
|
||||
}
|
||||
|
||||
private:
|
||||
float frame_period_; // Frame period
|
||||
float frame_duration_; // Window size
|
||||
float onset_window_; // Interval scanned for onset activity
|
||||
float speech_on_window_; // Inverval scanned for ongoing speech
|
||||
float offset_window_; // Interval scanned for offset evidence
|
||||
float offset_confirm_dur_; // Silence duration required to confirm offset
|
||||
float decision_threshold_; // Initial rms detection threshold
|
||||
float min_decision_threshold_; // Minimum rms detection threshold
|
||||
float fast_update_dur_; // Period for initial estimation of levels.
|
||||
float sample_rate_; // Expected sample rate.
|
||||
|
||||
// Time to add on either side of endpoint threshold crossings
|
||||
float endpoint_margin_;
|
||||
// Total dur within onset_window required to enter ONSET state
|
||||
float onset_detect_dur_;
|
||||
// Total on time within onset_window required to enter SPEECH_ON state
|
||||
float onset_confirm_dur_;
|
||||
// Minimum dur in SPEECH_ON state required to maintain ON state
|
||||
float on_maintain_dur_;
|
||||
// Minimum fundamental frequency for autocorrelation.
|
||||
float min_fundamental_frequency_;
|
||||
// Maximum fundamental frequency for autocorrelation.
|
||||
float max_fundamental_frequency_;
|
||||
// Period after start of user input that above threshold values are ignored.
|
||||
// This is to reject audio feedback contamination.
|
||||
float contamination_rejection_period_;
|
||||
};
|
||||
|
||||
} // namespace mozilla
|
||||
|
||||
#endif // CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_PARAMS_H_
|
||||
133449
dom/media/webspeech/recognition/models/dict/en-US.dic
Normal file
133449
dom/media/webspeech/recognition/models/dict/en-US.dic
Normal file
File diff suppressed because it is too large
Load diff
BIN
dom/media/webspeech/recognition/models/dict/en-US.dic.dmp
Normal file
BIN
dom/media/webspeech/recognition/models/dict/en-US.dic.dmp
Normal file
Binary file not shown.
34
dom/media/webspeech/recognition/models/en-US/README
Normal file
34
dom/media/webspeech/recognition/models/en-US/README
Normal file
|
|
@ -0,0 +1,34 @@
|
|||
/* ====================================================================
|
||||
* Copyright (c) 2015 Alpha Cephei Inc. All rights
|
||||
* reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in
|
||||
* the documentation and/or other materials provided with the
|
||||
* distribution.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY ALPHA CEPHEI INC. ``AS IS'' AND.
|
||||
* ANY EXPRESSED OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO,.
|
||||
* THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL ALPHA CEPHEI INC.
|
||||
* NOR ITS EMPLOYEES BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
||||
* SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT.
|
||||
* LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,.
|
||||
* DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY.
|
||||
* THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT.
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE.
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
* ====================================================================
|
||||
*
|
||||
*/
|
||||
|
||||
This directory contains generic US english acoustic model trained with
|
||||
latest sphinxtrain.
|
||||
12
dom/media/webspeech/recognition/models/en-US/feat.params
Normal file
12
dom/media/webspeech/recognition/models/en-US/feat.params
Normal file
|
|
@ -0,0 +1,12 @@
|
|||
-lowerf 130
|
||||
-upperf 3700
|
||||
-nfilt 20
|
||||
-transform dct
|
||||
-lifter 22
|
||||
-feat 1s_c_d_dd
|
||||
-svspec 0-12/13-25/26-38
|
||||
-agc none
|
||||
-cmn current
|
||||
-varnorm no
|
||||
-cmninit 30.09,0.49,5.51,7.43,-2.39,-1.64,-4.71,0.03,-5.11,0.20,4.84,6.14,0.35
|
||||
-model ptm
|
||||
137105
dom/media/webspeech/recognition/models/en-US/mdef
Normal file
137105
dom/media/webspeech/recognition/models/en-US/mdef
Normal file
File diff suppressed because it is too large
Load diff
BIN
dom/media/webspeech/recognition/models/en-US/means
Normal file
BIN
dom/media/webspeech/recognition/models/en-US/means
Normal file
Binary file not shown.
BIN
dom/media/webspeech/recognition/models/en-US/mixture_weights
Normal file
BIN
dom/media/webspeech/recognition/models/en-US/mixture_weights
Normal file
Binary file not shown.
5
dom/media/webspeech/recognition/models/en-US/noisedict
Normal file
5
dom/media/webspeech/recognition/models/en-US/noisedict
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
<s> SIL
|
||||
</s> SIL
|
||||
<sil> SIL
|
||||
[NOISE] +NSN+
|
||||
[SPEECH] +SPN+
|
||||
BIN
dom/media/webspeech/recognition/models/en-US/sendump
Normal file
BIN
dom/media/webspeech/recognition/models/en-US/sendump
Normal file
Binary file not shown.
BIN
dom/media/webspeech/recognition/models/en-US/transition_matrices
Normal file
BIN
dom/media/webspeech/recognition/models/en-US/transition_matrices
Normal file
Binary file not shown.
BIN
dom/media/webspeech/recognition/models/en-US/variances
Normal file
BIN
dom/media/webspeech/recognition/models/en-US/variances
Normal file
Binary file not shown.
88
dom/media/webspeech/recognition/moz.build
Normal file
88
dom/media/webspeech/recognition/moz.build
Normal file
|
|
@ -0,0 +1,88 @@
|
|||
# vim: set filetype=python:
|
||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
MOCHITEST_MANIFESTS += ['test/mochitest.ini']
|
||||
|
||||
XPIDL_MODULE = 'dom_webspeechrecognition'
|
||||
|
||||
XPIDL_SOURCES = [
|
||||
'nsISpeechRecognitionService.idl'
|
||||
]
|
||||
|
||||
EXPORTS.mozilla.dom += [
|
||||
'SpeechGrammar.h',
|
||||
'SpeechGrammarList.h',
|
||||
'SpeechRecognition.h',
|
||||
'SpeechRecognitionAlternative.h',
|
||||
'SpeechRecognitionResult.h',
|
||||
'SpeechRecognitionResultList.h',
|
||||
'SpeechStreamListener.h',
|
||||
]
|
||||
|
||||
if CONFIG['MOZ_WEBSPEECH_TEST_BACKEND']:
|
||||
EXPORTS.mozilla.dom += [
|
||||
'test/FakeSpeechRecognitionService.h',
|
||||
]
|
||||
|
||||
if CONFIG['MOZ_WEBSPEECH_POCKETSPHINX']:
|
||||
EXPORTS.mozilla.dom += [
|
||||
'PocketSphinxSpeechRecognitionService.h',
|
||||
]
|
||||
|
||||
UNIFIED_SOURCES += [
|
||||
'endpointer.cc',
|
||||
'energy_endpointer.cc',
|
||||
'energy_endpointer_params.cc',
|
||||
'SpeechGrammar.cpp',
|
||||
'SpeechGrammarList.cpp',
|
||||
'SpeechRecognition.cpp',
|
||||
'SpeechRecognitionAlternative.cpp',
|
||||
'SpeechRecognitionResult.cpp',
|
||||
'SpeechRecognitionResultList.cpp',
|
||||
'SpeechStreamListener.cpp',
|
||||
]
|
||||
|
||||
if CONFIG['MOZ_WEBSPEECH_TEST_BACKEND']:
|
||||
UNIFIED_SOURCES += [
|
||||
'test/FakeSpeechRecognitionService.cpp',
|
||||
]
|
||||
|
||||
if CONFIG['MOZ_WEBSPEECH_POCKETSPHINX']:
|
||||
UNIFIED_SOURCES += [
|
||||
'PocketSphinxSpeechRecognitionService.cpp',
|
||||
]
|
||||
|
||||
LOCAL_INCLUDES += [
|
||||
'/dom/base',
|
||||
'/media/sphinxbase',
|
||||
]
|
||||
|
||||
if CONFIG['MOZ_WEBSPEECH_POCKETSPHINX']:
|
||||
LOCAL_INCLUDES += [
|
||||
'/media/pocketsphinx',
|
||||
]
|
||||
|
||||
if CONFIG['MOZ_WEBSPEECH_MODELS']:
|
||||
FINAL_TARGET_FILES.models.dict += [
|
||||
'models/dict/en-US.dic',
|
||||
'models/dict/en-US.dic.dmp',
|
||||
]
|
||||
FINAL_TARGET_FILES.models['en-US'] += [
|
||||
'models/en-US/feat.params',
|
||||
'models/en-US/mdef',
|
||||
'models/en-US/means',
|
||||
'models/en-US/mixture_weights',
|
||||
'models/en-US/noisedict',
|
||||
'models/en-US/sendump',
|
||||
'models/en-US/transition_matrices',
|
||||
'models/en-US/variances',
|
||||
]
|
||||
|
||||
include('/ipc/chromium/chromium-config.mozbuild')
|
||||
|
||||
FINAL_LIBRARY = 'xul'
|
||||
|
||||
if CONFIG['GNU_CXX']:
|
||||
CXXFLAGS += ['-Wno-error=shadow']
|
||||
|
|
@ -0,0 +1,43 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 4 -*- */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsISupports.idl"
|
||||
|
||||
%{C++
|
||||
#include "mozilla/WeakPtr.h"
|
||||
|
||||
namespace mozilla {
|
||||
class AudioSegment;
|
||||
namespace dom {
|
||||
class SpeechRecognition;
|
||||
class SpeechRecognitionResultList;
|
||||
class SpeechGrammarList;
|
||||
class SpeechGrammar;
|
||||
}
|
||||
}
|
||||
%}
|
||||
|
||||
native SpeechRecognitionWeakPtr(mozilla::WeakPtr<mozilla::dom::SpeechRecognition>);
|
||||
[ptr] native AudioSegmentPtr(mozilla::AudioSegment);
|
||||
[ptr] native SpeechGrammarPtr(mozilla::dom::SpeechGrammar);
|
||||
[ptr] native SpeechGrammarListPtr(mozilla::dom::SpeechGrammarList);
|
||||
|
||||
[uuid(6fcb6ee8-a6db-49ba-9f06-355d7ee18ea7)]
|
||||
interface nsISpeechGrammarCompilationCallback : nsISupports {
|
||||
void grammarCompilationEnd(in SpeechGrammarPtr grammarObject, in boolean success);
|
||||
};
|
||||
|
||||
[uuid(8e97f287-f322-44e8-8888-8344fa408ef8)]
|
||||
interface nsISpeechRecognitionService : nsISupports {
|
||||
void initialize(in SpeechRecognitionWeakPtr aSpeechRecognition);
|
||||
void processAudioSegment(in AudioSegmentPtr aAudioSegment, in long aSampleRate);
|
||||
void validateAndSetGrammarList(in SpeechGrammarPtr aSpeechGrammar, in nsISpeechGrammarCompilationCallback aCallback);
|
||||
void soundEnd();
|
||||
void abort();
|
||||
};
|
||||
|
||||
%{C++
|
||||
#define NS_SPEECH_RECOGNITION_SERVICE_CONTRACTID_PREFIX "@mozilla.org/webspeech/service;1?name="
|
||||
%}
|
||||
|
|
@ -0,0 +1,119 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsThreadUtils.h"
|
||||
|
||||
#include "FakeSpeechRecognitionService.h"
|
||||
#include "MediaPrefs.h"
|
||||
|
||||
#include "SpeechRecognition.h"
|
||||
#include "SpeechRecognitionAlternative.h"
|
||||
#include "SpeechRecognitionResult.h"
|
||||
#include "SpeechRecognitionResultList.h"
|
||||
#include "nsIObserverService.h"
|
||||
#include "mozilla/Services.h"
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
using namespace dom;
|
||||
|
||||
NS_IMPL_ISUPPORTS(FakeSpeechRecognitionService, nsISpeechRecognitionService, nsIObserver)
|
||||
|
||||
FakeSpeechRecognitionService::FakeSpeechRecognitionService()
|
||||
{
|
||||
}
|
||||
|
||||
FakeSpeechRecognitionService::~FakeSpeechRecognitionService()
|
||||
{
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
FakeSpeechRecognitionService::Initialize(WeakPtr<SpeechRecognition> aSpeechRecognition)
|
||||
{
|
||||
mRecognition = aSpeechRecognition;
|
||||
nsCOMPtr<nsIObserverService> obs = services::GetObserverService();
|
||||
obs->AddObserver(this, SPEECH_RECOGNITION_TEST_EVENT_REQUEST_TOPIC, false);
|
||||
obs->AddObserver(this, SPEECH_RECOGNITION_TEST_END_TOPIC, false);
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
FakeSpeechRecognitionService::ProcessAudioSegment(AudioSegment* aAudioSegment, int32_t aSampleRate)
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
FakeSpeechRecognitionService::SoundEnd()
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
FakeSpeechRecognitionService::ValidateAndSetGrammarList(mozilla::dom::SpeechGrammar*, nsISpeechGrammarCompilationCallback*)
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
FakeSpeechRecognitionService::Abort()
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
FakeSpeechRecognitionService::Observe(nsISupports* aSubject, const char* aTopic, const char16_t* aData)
|
||||
{
|
||||
MOZ_ASSERT(MediaPrefs::WebSpeechFakeRecognitionService(),
|
||||
"Got request to fake recognition service event, but "
|
||||
TEST_PREFERENCE_FAKE_RECOGNITION_SERVICE " is not set");
|
||||
|
||||
if (!strcmp(aTopic, SPEECH_RECOGNITION_TEST_END_TOPIC)) {
|
||||
nsCOMPtr<nsIObserverService> obs = services::GetObserverService();
|
||||
obs->RemoveObserver(this, SPEECH_RECOGNITION_TEST_EVENT_REQUEST_TOPIC);
|
||||
obs->RemoveObserver(this, SPEECH_RECOGNITION_TEST_END_TOPIC);
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
const nsDependentString eventName = nsDependentString(aData);
|
||||
|
||||
if (eventName.EqualsLiteral("EVENT_RECOGNITIONSERVICE_ERROR")) {
|
||||
mRecognition->DispatchError(SpeechRecognition::EVENT_RECOGNITIONSERVICE_ERROR,
|
||||
SpeechRecognitionErrorCode::Network, // TODO different codes?
|
||||
NS_LITERAL_STRING("RECOGNITIONSERVICE_ERROR test event"));
|
||||
|
||||
} else if (eventName.EqualsLiteral("EVENT_RECOGNITIONSERVICE_FINAL_RESULT")) {
|
||||
RefPtr<SpeechEvent> event =
|
||||
new SpeechEvent(mRecognition,
|
||||
SpeechRecognition::EVENT_RECOGNITIONSERVICE_FINAL_RESULT);
|
||||
|
||||
event->mRecognitionResultList = BuildMockResultList();
|
||||
NS_DispatchToMainThread(event);
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
SpeechRecognitionResultList*
|
||||
FakeSpeechRecognitionService::BuildMockResultList()
|
||||
{
|
||||
SpeechRecognitionResultList* resultList = new SpeechRecognitionResultList(mRecognition);
|
||||
SpeechRecognitionResult* result = new SpeechRecognitionResult(mRecognition);
|
||||
if (0 < mRecognition->MaxAlternatives()) {
|
||||
SpeechRecognitionAlternative* alternative = new SpeechRecognitionAlternative(mRecognition);
|
||||
|
||||
alternative->mTranscript = NS_LITERAL_STRING("Mock final result");
|
||||
alternative->mConfidence = 0.0f;
|
||||
|
||||
result->mItems.AppendElement(alternative);
|
||||
}
|
||||
resultList->mItems.AppendElement(result);
|
||||
|
||||
return resultList;
|
||||
}
|
||||
|
||||
} // namespace mozilla
|
||||
|
|
@ -0,0 +1,38 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_FakeSpeechRecognitionService_h
|
||||
#define mozilla_dom_FakeSpeechRecognitionService_h
|
||||
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsIObserver.h"
|
||||
#include "nsISpeechRecognitionService.h"
|
||||
|
||||
#define NS_FAKE_SPEECH_RECOGNITION_SERVICE_CID \
|
||||
{0x48c345e7, 0x9929, 0x4f9a, {0xa5, 0x63, 0xf4, 0x78, 0x22, 0x2d, 0xab, 0xcd}};
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
class FakeSpeechRecognitionService : public nsISpeechRecognitionService,
|
||||
public nsIObserver
|
||||
{
|
||||
public:
|
||||
NS_DECL_ISUPPORTS
|
||||
NS_DECL_NSISPEECHRECOGNITIONSERVICE
|
||||
NS_DECL_NSIOBSERVER
|
||||
|
||||
FakeSpeechRecognitionService();
|
||||
|
||||
private:
|
||||
virtual ~FakeSpeechRecognitionService();
|
||||
|
||||
WeakPtr<dom::SpeechRecognition> mRecognition;
|
||||
dom::SpeechRecognitionResultList* BuildMockResultList();
|
||||
};
|
||||
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
181
dom/media/webspeech/recognition/test/head.js
Normal file
181
dom/media/webspeech/recognition/test/head.js
Normal file
|
|
@ -0,0 +1,181 @@
|
|||
"use strict";
|
||||
|
||||
const DEFAULT_AUDIO_SAMPLE_FILE = "hello.ogg";
|
||||
const SPEECH_RECOGNITION_TEST_REQUEST_EVENT_TOPIC = "SpeechRecognitionTest:RequestEvent";
|
||||
const SPEECH_RECOGNITION_TEST_END_TOPIC = "SpeechRecognitionTest:End";
|
||||
|
||||
var errorCodes = {
|
||||
NO_SPEECH : "no-speech",
|
||||
ABORTED : "aborted",
|
||||
AUDIO_CAPTURE : "audio-capture",
|
||||
NETWORK : "network",
|
||||
NOT_ALLOWED : "not-allowed",
|
||||
SERVICE_NOT_ALLOWED : "service-not-allowed",
|
||||
BAD_GRAMMAR : "bad-grammar",
|
||||
LANGUAGE_NOT_SUPPORTED : "language-not-supported"
|
||||
};
|
||||
|
||||
var Services = SpecialPowers.Cu.import("resource://gre/modules/Services.jsm").Services;
|
||||
|
||||
function EventManager(sr) {
|
||||
var self = this;
|
||||
var nEventsExpected = 0;
|
||||
self.eventsReceived = [];
|
||||
|
||||
var allEvents = [
|
||||
"audiostart",
|
||||
"soundstart",
|
||||
"speechstart",
|
||||
"speechend",
|
||||
"soundend",
|
||||
"audioend",
|
||||
"result",
|
||||
"nomatch",
|
||||
"error",
|
||||
"start",
|
||||
"end"
|
||||
];
|
||||
|
||||
var eventDependencies = {
|
||||
"speechend": "speechstart",
|
||||
"soundend": "soundstart",
|
||||
"audioend": "audiostart"
|
||||
};
|
||||
|
||||
var isDone = false;
|
||||
|
||||
// set up grammar
|
||||
var sgl = new SpeechGrammarList();
|
||||
sgl.addFromString("#JSGF V1.0; grammar test; public <simple> = hello ;", 1);
|
||||
sr.grammars = sgl;
|
||||
|
||||
// AUDIO_DATA events are asynchronous,
|
||||
// so we queue events requested while they are being
|
||||
// issued to make them seem synchronous
|
||||
var isSendingAudioData = false;
|
||||
var queuedEventRequests = [];
|
||||
|
||||
// register default handlers
|
||||
for (var i = 0; i < allEvents.length; i++) {
|
||||
(function (eventName) {
|
||||
sr["on" + eventName] = function (evt) {
|
||||
var message = "unexpected event: " + eventName;
|
||||
if (eventName == "error") {
|
||||
message += " -- " + evt.message;
|
||||
}
|
||||
|
||||
ok(false, message);
|
||||
if (self.doneFunc && !isDone) {
|
||||
isDone = true;
|
||||
self.doneFunc();
|
||||
}
|
||||
};
|
||||
})(allEvents[i]);
|
||||
}
|
||||
|
||||
self.expect = function EventManager_expect(eventName, cb) {
|
||||
nEventsExpected++;
|
||||
|
||||
sr["on" + eventName] = function(evt) {
|
||||
self.eventsReceived.push(eventName);
|
||||
ok(true, "received event " + eventName);
|
||||
|
||||
var dep = eventDependencies[eventName];
|
||||
if (dep) {
|
||||
ok(self.eventsReceived.indexOf(dep) >= 0,
|
||||
eventName + " must come after " + dep);
|
||||
}
|
||||
|
||||
cb && cb(evt, sr);
|
||||
if (self.doneFunc && !isDone &&
|
||||
nEventsExpected === self.eventsReceived.length) {
|
||||
isDone = true;
|
||||
self.doneFunc();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
self.start = function EventManager_start() {
|
||||
isSendingAudioData = true;
|
||||
var audioTag = document.createElement("audio");
|
||||
audioTag.src = self.audioSampleFile;
|
||||
|
||||
var stream = audioTag.mozCaptureStreamUntilEnded();
|
||||
audioTag.addEventListener("ended", function() {
|
||||
info("Sample stream ended, requesting queued events");
|
||||
isSendingAudioData = false;
|
||||
while (queuedEventRequests.length) {
|
||||
self.requestFSMEvent(queuedEventRequests.shift());
|
||||
}
|
||||
});
|
||||
|
||||
audioTag.play();
|
||||
sr.start(stream);
|
||||
}
|
||||
|
||||
self.requestFSMEvent = function EventManager_requestFSMEvent(eventName) {
|
||||
if (isSendingAudioData) {
|
||||
info("Queuing event " + eventName + " until we're done sending audio data");
|
||||
queuedEventRequests.push(eventName);
|
||||
return;
|
||||
}
|
||||
|
||||
info("requesting " + eventName);
|
||||
Services.obs.notifyObservers(null,
|
||||
SPEECH_RECOGNITION_TEST_REQUEST_EVENT_TOPIC,
|
||||
eventName);
|
||||
}
|
||||
|
||||
self.requestTestEnd = function EventManager_requestTestEnd() {
|
||||
Services.obs.notifyObservers(null, SPEECH_RECOGNITION_TEST_END_TOPIC, null);
|
||||
}
|
||||
}
|
||||
|
||||
function buildResultCallback(transcript) {
|
||||
return (function(evt) {
|
||||
is(evt.results[0][0].transcript, transcript, "expect correct transcript");
|
||||
});
|
||||
}
|
||||
|
||||
function buildErrorCallback(errcode) {
|
||||
return (function(err) {
|
||||
is(err.error, errcode, "expect correct error code");
|
||||
});
|
||||
}
|
||||
|
||||
function performTest(options) {
|
||||
var prefs = options.prefs;
|
||||
|
||||
prefs.unshift(
|
||||
["media.webspeech.recognition.enable", true],
|
||||
["media.webspeech.test.enable", true]
|
||||
);
|
||||
|
||||
SpecialPowers.pushPrefEnv({set: prefs}, function() {
|
||||
var sr = new SpeechRecognition();
|
||||
var em = new EventManager(sr);
|
||||
|
||||
for (var eventName in options.expectedEvents) {
|
||||
var cb = options.expectedEvents[eventName];
|
||||
em.expect(eventName, cb);
|
||||
}
|
||||
|
||||
em.doneFunc = function() {
|
||||
em.requestTestEnd();
|
||||
if (options.doneFunc) {
|
||||
options.doneFunc();
|
||||
}
|
||||
}
|
||||
|
||||
em.audioSampleFile = DEFAULT_AUDIO_SAMPLE_FILE;
|
||||
if (options.audioSampleFile) {
|
||||
em.audioSampleFile = options.audioSampleFile;
|
||||
}
|
||||
|
||||
em.start();
|
||||
|
||||
for (var i = 0; i < options.eventsToRequest.length; i++) {
|
||||
em.requestFSMEvent(options.eventsToRequest[i]);
|
||||
}
|
||||
});
|
||||
}
|
||||
BIN
dom/media/webspeech/recognition/test/hello.ogg
Normal file
BIN
dom/media/webspeech/recognition/test/hello.ogg
Normal file
Binary file not shown.
1
dom/media/webspeech/recognition/test/hello.ogg^headers^
Normal file
1
dom/media/webspeech/recognition/test/hello.ogg^headers^
Normal file
|
|
@ -0,0 +1 @@
|
|||
Cache-Control: no-store
|
||||
21
dom/media/webspeech/recognition/test/mochitest.ini
Normal file
21
dom/media/webspeech/recognition/test/mochitest.ini
Normal file
|
|
@ -0,0 +1,21 @@
|
|||
[DEFAULT]
|
||||
tags=msg
|
||||
subsuite = media
|
||||
support-files =
|
||||
head.js
|
||||
hello.ogg
|
||||
hello.ogg^headers^
|
||||
silence.ogg
|
||||
silence.ogg^headers^
|
||||
[test_abort.html]
|
||||
skip-if = toolkit == 'android' # bug 1037287
|
||||
[test_audio_capture_error.html]
|
||||
[test_call_start_from_end_handler.html]
|
||||
tags=capturestream
|
||||
skip-if = (android_version == '18' && debug) # bug 967606
|
||||
[test_nested_eventloop.html]
|
||||
skip-if = toolkit == 'android'
|
||||
[test_preference_enable.html]
|
||||
[test_recognition_service_error.html]
|
||||
[test_success_without_recognition_service.html]
|
||||
[test_timeout.html]
|
||||
BIN
dom/media/webspeech/recognition/test/silence.ogg
Normal file
BIN
dom/media/webspeech/recognition/test/silence.ogg
Normal file
Binary file not shown.
|
|
@ -0,0 +1 @@
|
|||
Cache-Control: no-store
|
||||
71
dom/media/webspeech/recognition/test/test_abort.html
Normal file
71
dom/media/webspeech/recognition/test/test_abort.html
Normal file
|
|
@ -0,0 +1,71 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 650295 -- Call abort from inside handlers</title>
|
||||
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
|
||||
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
|
||||
<script type="application/javascript" src="head.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="text/javascript">
|
||||
SimpleTest.waitForExplicitFinish();
|
||||
|
||||
// Abort inside event handlers, should't get a
|
||||
// result after that
|
||||
|
||||
var nextEventIdx = 0;
|
||||
var eventsToAbortOn = [
|
||||
"start",
|
||||
"audiostart",
|
||||
"speechstart",
|
||||
"speechend",
|
||||
"audioend"
|
||||
];
|
||||
|
||||
function doNextTest() {
|
||||
var nextEvent = eventsToAbortOn[nextEventIdx];
|
||||
var expectedEvents = {
|
||||
"start": null,
|
||||
"audiostart": null,
|
||||
"audioend": null,
|
||||
"end": null
|
||||
};
|
||||
|
||||
if (nextEventIdx >= eventsToAbortOn.indexOf("speechstart")) {
|
||||
expectedEvents["speechstart"] = null;
|
||||
}
|
||||
|
||||
if (nextEventIdx >= eventsToAbortOn.indexOf("speechend")) {
|
||||
expectedEvents["speechend"] = null;
|
||||
}
|
||||
|
||||
info("Aborting on " + nextEvent);
|
||||
expectedEvents[nextEvent] = function(evt, sr) {
|
||||
sr.abort();
|
||||
};
|
||||
|
||||
nextEventIdx++;
|
||||
|
||||
performTest({
|
||||
eventsToRequest: [],
|
||||
expectedEvents: expectedEvents,
|
||||
doneFunc: (nextEventIdx < eventsToAbortOn.length) ? doNextTest : SimpleTest.finish,
|
||||
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
|
||||
});
|
||||
}
|
||||
|
||||
doNextTest();
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
|
|
@ -0,0 +1,40 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 650295 -- Behavior on audio error</title>
|
||||
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
|
||||
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
|
||||
<script type="application/javascript" src="head.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="text/javascript">
|
||||
SimpleTest.waitForExplicitFinish();
|
||||
|
||||
performTest({
|
||||
eventsToRequest: ['EVENT_AUDIO_ERROR'],
|
||||
expectedEvents: {
|
||||
'start': null,
|
||||
'audiostart': null,
|
||||
'speechstart': null,
|
||||
'speechend': null,
|
||||
'audioend': null,
|
||||
'error': buildErrorCallback(errorCodes.AUDIO_CAPTURE),
|
||||
'end': null
|
||||
},
|
||||
doneFunc: SimpleTest.finish,
|
||||
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
|
||||
});
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
|
|
@ -0,0 +1,100 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 650295 -- Restart recognition from end handler</title>
|
||||
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
|
||||
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
|
||||
<script type="application/javascript" src="head.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="text/javascript">
|
||||
SimpleTest.waitForExplicitFinish();
|
||||
|
||||
function createAudioStream() {
|
||||
var audioTag = document.createElement("audio");
|
||||
audioTag.src = DEFAULT_AUDIO_SAMPLE_FILE;
|
||||
|
||||
var stream = audioTag.mozCaptureStreamUntilEnded();
|
||||
audioTag.play();
|
||||
|
||||
return stream;
|
||||
}
|
||||
|
||||
var done = false;
|
||||
function endHandler(evt, sr) {
|
||||
if (done) {
|
||||
SimpleTest.finish();
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
var stream = createAudioStream();
|
||||
sr.start(stream); // shouldn't fail
|
||||
} catch (err) {
|
||||
ok(false, "Failed to start() from end() callback");
|
||||
}
|
||||
|
||||
// calling start() may cause some callbacks to fire, but we're
|
||||
// no longer interested in them, except for onend, which is where
|
||||
// we'll conclude the test.
|
||||
sr.onstart = null;
|
||||
sr.onaudiostart = null;
|
||||
sr.onspeechstart = null;
|
||||
sr.onspeechend = null;
|
||||
sr.onaudioend = null;
|
||||
sr.onresult = null;
|
||||
|
||||
// FIXME(ggp) the state transition caused by start() is async,
|
||||
// but abort() is sync (see bug 1055093). until we normalize
|
||||
// state transitions, we need to setTimeout here to make sure
|
||||
// abort() finds the speech recognition object in the correct
|
||||
// state (namely, STATE_STARTING).
|
||||
setTimeout(function() {
|
||||
sr.abort();
|
||||
done = true;
|
||||
});
|
||||
|
||||
info("Successfully start() from end() callback");
|
||||
}
|
||||
|
||||
function expectExceptionHandler(evt, sr) {
|
||||
try {
|
||||
sr.start(createAudioStream());
|
||||
} catch (err) {
|
||||
is(err.name, "InvalidStateError");
|
||||
return;
|
||||
}
|
||||
|
||||
ok(false, "Calling start() didn't raise InvalidStateError");
|
||||
}
|
||||
|
||||
performTest({
|
||||
eventsToRequest: [
|
||||
'EVENT_RECOGNITIONSERVICE_FINAL_RESULT'
|
||||
],
|
||||
expectedEvents: {
|
||||
'start': expectExceptionHandler,
|
||||
'audiostart': expectExceptionHandler,
|
||||
'speechstart': expectExceptionHandler,
|
||||
'speechend': expectExceptionHandler,
|
||||
'audioend': expectExceptionHandler,
|
||||
'result': buildResultCallback("Mock final result"),
|
||||
'end': endHandler,
|
||||
},
|
||||
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
|
||||
});
|
||||
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
|
|
@ -0,0 +1,81 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 650295 -- Spin the event loop from inside a callback</title>
|
||||
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
|
||||
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
|
||||
<script type="application/javascript" src="head.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="text/javascript">
|
||||
SimpleTest.waitForExplicitFinish();
|
||||
|
||||
/*
|
||||
* window.showModalDialog() can be used to spin the event loop, causing
|
||||
* queued SpeechEvents (such as those created by calls to start(), stop()
|
||||
* or abort()) to be processed immediately.
|
||||
* When this is done from inside DOM event handlers, it is possible to
|
||||
* cause reentrancy in our C++ code, which we should be able to withstand.
|
||||
*/
|
||||
function abortAndSpinEventLoop(evt, sr) {
|
||||
sr.abort();
|
||||
SpecialPowers.spinEventLoop(window);
|
||||
}
|
||||
function doneFunc() {
|
||||
// Trigger gc now and wait some time to make sure this test gets the blame
|
||||
// for any assertions caused by showModalDialog
|
||||
//
|
||||
// NB - The assertions should be gone, but this looks too scary to touch
|
||||
// during batch cleanup.
|
||||
var count = 0, GC_COUNT = 4;
|
||||
|
||||
function triggerGCOrFinish() {
|
||||
SpecialPowers.gc();
|
||||
count++;
|
||||
|
||||
if (count == GC_COUNT) {
|
||||
SimpleTest.finish();
|
||||
}
|
||||
}
|
||||
|
||||
for (var i = 0; i < GC_COUNT; i++) {
|
||||
setTimeout(triggerGCOrFinish, 0);
|
||||
}
|
||||
}
|
||||
|
||||
/*
|
||||
* We start by performing a normal start, then abort from the audiostart
|
||||
* callback and force the EVENT_ABORT to be processed while still inside
|
||||
* the event handler. This causes the recording to stop, which raises
|
||||
* the audioend and (later on) end events.
|
||||
* Then, we abort (once again spinning the event loop) from the audioend
|
||||
* handler, attempting to cause a re-entry into the abort code. This second
|
||||
* call should be ignored, and we get the end callback and finish.
|
||||
*/
|
||||
|
||||
performTest({
|
||||
eventsToRequest: [],
|
||||
expectedEvents: {
|
||||
"audiostart": abortAndSpinEventLoop,
|
||||
"audioend": abortAndSpinEventLoop,
|
||||
"end": null
|
||||
},
|
||||
doneFunc: doneFunc,
|
||||
prefs: [["media.webspeech.test.fake_fsm_events", true],
|
||||
["media.webspeech.test.fake_recognition_service", true]]
|
||||
});
|
||||
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
|
|
@ -0,0 +1,43 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 650295 -- No objects should be visible with preference disabled</title>
|
||||
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
|
||||
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="text/javascript">
|
||||
SimpleTest.waitForExplicitFinish();
|
||||
|
||||
SpecialPowers.pushPrefEnv({
|
||||
set: [["media.webspeech.recognition.enable", false]]
|
||||
}, function() {
|
||||
var objects = [
|
||||
"SpeechRecognition",
|
||||
"SpeechGrammar",
|
||||
"SpeechRecognitionResult",
|
||||
"SpeechRecognitionResultList",
|
||||
"SpeechRecognitionAlternative"
|
||||
];
|
||||
|
||||
for (var i = 0; i < objects.length; i++) {
|
||||
is(window[objects[i]], undefined,
|
||||
objects[i] + " should be undefined with pref off");
|
||||
}
|
||||
|
||||
SimpleTest.finish();
|
||||
});
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
|
|
@ -0,0 +1,43 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 650295 -- Behavior on recognition service error</title>
|
||||
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
|
||||
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
|
||||
<script type="application/javascript" src="head.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="text/javascript">
|
||||
SimpleTest.waitForExplicitFinish();
|
||||
|
||||
performTest({
|
||||
eventsToRequest: [
|
||||
'EVENT_RECOGNITIONSERVICE_ERROR'
|
||||
],
|
||||
expectedEvents: {
|
||||
'start': null,
|
||||
'audiostart': null,
|
||||
'speechstart': null,
|
||||
'speechend': null,
|
||||
'audioend': null,
|
||||
'error': buildErrorCallback(errorCodes.NETWORK),
|
||||
'end': null
|
||||
},
|
||||
doneFunc: SimpleTest.finish,
|
||||
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
|
||||
});
|
||||
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
|
|
@ -0,0 +1,43 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 650295 -- Success with fake recognition service</title>
|
||||
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
|
||||
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
|
||||
<script type="application/javascript" src="head.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="text/javascript">
|
||||
SimpleTest.waitForExplicitFinish();
|
||||
|
||||
performTest({
|
||||
eventsToRequest: [
|
||||
'EVENT_RECOGNITIONSERVICE_FINAL_RESULT'
|
||||
],
|
||||
expectedEvents: {
|
||||
'start': null,
|
||||
'audiostart': null,
|
||||
'speechstart': null,
|
||||
'speechend': null,
|
||||
'audioend': null,
|
||||
'result': buildResultCallback("Mock final result"),
|
||||
'end': null
|
||||
},
|
||||
doneFunc:SimpleTest.finish,
|
||||
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
|
||||
});
|
||||
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
40
dom/media/webspeech/recognition/test/test_timeout.html
Normal file
40
dom/media/webspeech/recognition/test/test_timeout.html
Normal file
|
|
@ -0,0 +1,40 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 650295 -- Timeout for user speech</title>
|
||||
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
|
||||
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
|
||||
<script type="application/javascript" src="head.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="text/javascript">
|
||||
SimpleTest.waitForExplicitFinish();
|
||||
|
||||
performTest({
|
||||
eventsToRequest: [],
|
||||
expectedEvents: {
|
||||
"start": null,
|
||||
"audiostart": null,
|
||||
"audioend": null,
|
||||
"error": buildErrorCallback(errorCodes.NO_SPEECH),
|
||||
"end": null
|
||||
},
|
||||
doneFunc: SimpleTest.finish,
|
||||
audioSampleFile: "silence.ogg",
|
||||
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
|
||||
});
|
||||
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
333
dom/media/webspeech/synth/SpeechSynthesis.cpp
Normal file
333
dom/media/webspeech/synth/SpeechSynthesis.cpp
Normal file
|
|
@ -0,0 +1,333 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsISupportsPrimitives.h"
|
||||
#include "nsSpeechTask.h"
|
||||
#include "mozilla/Logging.h"
|
||||
|
||||
#include "mozilla/dom/ContentChild.h"
|
||||
#include "mozilla/dom/Element.h"
|
||||
|
||||
#include "mozilla/dom/SpeechSynthesisBinding.h"
|
||||
#include "SpeechSynthesis.h"
|
||||
#include "nsSynthVoiceRegistry.h"
|
||||
#include "nsIDocument.h"
|
||||
|
||||
#undef LOG
|
||||
mozilla::LogModule*
|
||||
GetSpeechSynthLog()
|
||||
{
|
||||
static mozilla::LazyLogModule sLog("SpeechSynthesis");
|
||||
|
||||
return sLog;
|
||||
}
|
||||
#define LOG(type, msg) MOZ_LOG(GetSpeechSynthLog(), type, msg)
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION_CLASS(SpeechSynthesis)
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION_UNLINK_BEGIN_INHERITED(SpeechSynthesis, DOMEventTargetHelper)
|
||||
NS_IMPL_CYCLE_COLLECTION_UNLINK(mCurrentTask)
|
||||
NS_IMPL_CYCLE_COLLECTION_UNLINK(mSpeechQueue)
|
||||
tmp->mVoiceCache.Clear();
|
||||
NS_IMPL_CYCLE_COLLECTION_UNLINK_END
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION_TRAVERSE_BEGIN_INHERITED(SpeechSynthesis, DOMEventTargetHelper)
|
||||
NS_IMPL_CYCLE_COLLECTION_TRAVERSE(mCurrentTask)
|
||||
NS_IMPL_CYCLE_COLLECTION_TRAVERSE(mSpeechQueue)
|
||||
for (auto iter = tmp->mVoiceCache.Iter(); !iter.Done(); iter.Next()) {
|
||||
SpeechSynthesisVoice* voice = iter.UserData();
|
||||
cb.NoteXPCOMChild(voice);
|
||||
}
|
||||
NS_IMPL_CYCLE_COLLECTION_TRAVERSE_END
|
||||
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION_INHERITED(SpeechSynthesis)
|
||||
NS_INTERFACE_MAP_ENTRY(nsIObserver)
|
||||
NS_INTERFACE_MAP_ENTRY(nsISupportsWeakReference)
|
||||
NS_INTERFACE_MAP_END_INHERITING(DOMEventTargetHelper)
|
||||
|
||||
NS_IMPL_ADDREF_INHERITED(SpeechSynthesis, DOMEventTargetHelper)
|
||||
NS_IMPL_RELEASE_INHERITED(SpeechSynthesis, DOMEventTargetHelper)
|
||||
|
||||
SpeechSynthesis::SpeechSynthesis(nsPIDOMWindowInner* aParent)
|
||||
: DOMEventTargetHelper(aParent)
|
||||
, mHoldQueue(false)
|
||||
, mInnerID(aParent->WindowID())
|
||||
{
|
||||
MOZ_ASSERT(aParent->IsInnerWindow());
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
|
||||
nsCOMPtr<nsIObserverService> obs = mozilla::services::GetObserverService();
|
||||
if (obs) {
|
||||
obs->AddObserver(this, "inner-window-destroyed", true);
|
||||
obs->AddObserver(this, "synth-voices-changed", true);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
SpeechSynthesis::~SpeechSynthesis()
|
||||
{
|
||||
}
|
||||
|
||||
JSObject*
|
||||
SpeechSynthesis::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
|
||||
{
|
||||
return SpeechSynthesisBinding::Wrap(aCx, this, aGivenProto);
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesis::Pending() const
|
||||
{
|
||||
switch (mSpeechQueue.Length()) {
|
||||
case 0:
|
||||
return false;
|
||||
|
||||
case 1:
|
||||
return mSpeechQueue.ElementAt(0)->GetState() == SpeechSynthesisUtterance::STATE_PENDING;
|
||||
|
||||
default:
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesis::Speaking() const
|
||||
{
|
||||
if (!mSpeechQueue.IsEmpty() &&
|
||||
mSpeechQueue.ElementAt(0)->GetState() == SpeechSynthesisUtterance::STATE_SPEAKING) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Returns global speaking state if global queue is enabled. Or false.
|
||||
return nsSynthVoiceRegistry::GetInstance()->IsSpeaking();
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesis::Paused() const
|
||||
{
|
||||
return mHoldQueue || (mCurrentTask && mCurrentTask->IsPrePaused()) ||
|
||||
(!mSpeechQueue.IsEmpty() && mSpeechQueue.ElementAt(0)->IsPaused());
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesis::HasEmptyQueue() const
|
||||
{
|
||||
return mSpeechQueue.Length() == 0;
|
||||
}
|
||||
|
||||
bool SpeechSynthesis::HasVoices() const
|
||||
{
|
||||
uint32_t voiceCount = mVoiceCache.Count();
|
||||
if (voiceCount == 0) {
|
||||
nsresult rv = nsSynthVoiceRegistry::GetInstance()->GetVoiceCount(&voiceCount);
|
||||
if(NS_WARN_IF(NS_FAILED(rv))) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return voiceCount != 0;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesis::Speak(SpeechSynthesisUtterance& aUtterance)
|
||||
{
|
||||
if (aUtterance.mState != SpeechSynthesisUtterance::STATE_NONE) {
|
||||
// XXX: Should probably raise an error
|
||||
return;
|
||||
}
|
||||
|
||||
mSpeechQueue.AppendElement(&aUtterance);
|
||||
aUtterance.mState = SpeechSynthesisUtterance::STATE_PENDING;
|
||||
|
||||
// If we only have one item in the queue, we aren't pre-paused, and
|
||||
// we have voices available, speak it.
|
||||
if (mSpeechQueue.Length() == 1 && !mCurrentTask && !mHoldQueue && HasVoices()) {
|
||||
AdvanceQueue();
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesis::AdvanceQueue()
|
||||
{
|
||||
LOG(LogLevel::Debug,
|
||||
("SpeechSynthesis::AdvanceQueue length=%d", mSpeechQueue.Length()));
|
||||
|
||||
if (mSpeechQueue.IsEmpty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
RefPtr<SpeechSynthesisUtterance> utterance = mSpeechQueue.ElementAt(0);
|
||||
|
||||
nsAutoString docLang;
|
||||
nsCOMPtr<nsPIDOMWindowInner> window = GetOwner();
|
||||
nsIDocument* doc = window ? window->GetExtantDoc() : nullptr;
|
||||
|
||||
if (doc) {
|
||||
Element* elm = doc->GetHtmlElement();
|
||||
|
||||
if (elm) {
|
||||
elm->GetLang(docLang);
|
||||
}
|
||||
}
|
||||
|
||||
mCurrentTask =
|
||||
nsSynthVoiceRegistry::GetInstance()->SpeakUtterance(*utterance, docLang);
|
||||
|
||||
if (mCurrentTask) {
|
||||
mCurrentTask->SetSpeechSynthesis(this);
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesis::Cancel()
|
||||
{
|
||||
if (!mSpeechQueue.IsEmpty() &&
|
||||
mSpeechQueue.ElementAt(0)->GetState() == SpeechSynthesisUtterance::STATE_SPEAKING) {
|
||||
// Remove all queued utterances except for current one, we will remove it
|
||||
// in OnEnd
|
||||
mSpeechQueue.RemoveElementsAt(1, mSpeechQueue.Length() - 1);
|
||||
} else {
|
||||
mSpeechQueue.Clear();
|
||||
}
|
||||
|
||||
if (mCurrentTask) {
|
||||
mCurrentTask->Cancel();
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesis::Pause()
|
||||
{
|
||||
if (Paused()) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (mCurrentTask && !mSpeechQueue.IsEmpty() &&
|
||||
mSpeechQueue.ElementAt(0)->GetState() != SpeechSynthesisUtterance::STATE_ENDED) {
|
||||
mCurrentTask->Pause();
|
||||
} else {
|
||||
mHoldQueue = true;
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesis::Resume()
|
||||
{
|
||||
if (!Paused()) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (mCurrentTask) {
|
||||
mCurrentTask->Resume();
|
||||
} else {
|
||||
mHoldQueue = false;
|
||||
AdvanceQueue();
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesis::OnEnd(const nsSpeechTask* aTask)
|
||||
{
|
||||
MOZ_ASSERT(mCurrentTask == aTask);
|
||||
|
||||
if (!mSpeechQueue.IsEmpty()) {
|
||||
mSpeechQueue.RemoveElementAt(0);
|
||||
}
|
||||
|
||||
mCurrentTask = nullptr;
|
||||
AdvanceQueue();
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesis::GetVoices(nsTArray< RefPtr<SpeechSynthesisVoice> >& aResult)
|
||||
{
|
||||
aResult.Clear();
|
||||
uint32_t voiceCount = 0;
|
||||
|
||||
nsresult rv = nsSynthVoiceRegistry::GetInstance()->GetVoiceCount(&voiceCount);
|
||||
if(NS_WARN_IF(NS_FAILED(rv))) {
|
||||
return;
|
||||
}
|
||||
|
||||
nsISupports* voiceParent = NS_ISUPPORTS_CAST(nsIObserver*, this);
|
||||
|
||||
for (uint32_t i = 0; i < voiceCount; i++) {
|
||||
nsAutoString uri;
|
||||
rv = nsSynthVoiceRegistry::GetInstance()->GetVoice(i, uri);
|
||||
|
||||
if (NS_FAILED(rv)) {
|
||||
NS_WARNING("Failed to retrieve voice from registry");
|
||||
continue;
|
||||
}
|
||||
|
||||
SpeechSynthesisVoice* voice = mVoiceCache.GetWeak(uri);
|
||||
|
||||
if (!voice) {
|
||||
voice = new SpeechSynthesisVoice(voiceParent, uri);
|
||||
}
|
||||
|
||||
aResult.AppendElement(voice);
|
||||
}
|
||||
|
||||
mVoiceCache.Clear();
|
||||
|
||||
for (uint32_t i = 0; i < aResult.Length(); i++) {
|
||||
SpeechSynthesisVoice* voice = aResult[i];
|
||||
mVoiceCache.Put(voice->mUri, voice);
|
||||
}
|
||||
}
|
||||
|
||||
// For testing purposes, allows us to cancel the current task that is
|
||||
// misbehaving, and flush the queue.
|
||||
void
|
||||
SpeechSynthesis::ForceEnd()
|
||||
{
|
||||
if (mCurrentTask) {
|
||||
mCurrentTask->ForceEnd();
|
||||
}
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechSynthesis::Observe(nsISupports* aSubject, const char* aTopic,
|
||||
const char16_t* aData)
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
|
||||
|
||||
if (strcmp(aTopic, "inner-window-destroyed") == 0) {
|
||||
nsCOMPtr<nsISupportsPRUint64> wrapper = do_QueryInterface(aSubject);
|
||||
NS_ENSURE_TRUE(wrapper, NS_ERROR_FAILURE);
|
||||
|
||||
uint64_t innerID;
|
||||
nsresult rv = wrapper->GetData(&innerID);
|
||||
NS_ENSURE_SUCCESS(rv, rv);
|
||||
|
||||
if (innerID == mInnerID) {
|
||||
Cancel();
|
||||
|
||||
nsCOMPtr<nsIObserverService> obs = mozilla::services::GetObserverService();
|
||||
if (obs) {
|
||||
obs->RemoveObserver(this, "inner-window-destroyed");
|
||||
}
|
||||
}
|
||||
} else if (strcmp(aTopic, "synth-voices-changed") == 0) {
|
||||
LOG(LogLevel::Debug, ("SpeechSynthesis::onvoiceschanged"));
|
||||
DispatchTrustedEvent(NS_LITERAL_STRING("voiceschanged"));
|
||||
// If we have a pending item, and voices become available, speak it.
|
||||
if (!mCurrentTask && !mHoldQueue && HasVoices()) {
|
||||
AdvanceQueue();
|
||||
}
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
85
dom/media/webspeech/synth/SpeechSynthesis.h
Normal file
85
dom/media/webspeech/synth/SpeechSynthesis.h
Normal file
|
|
@ -0,0 +1,85 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechSynthesis_h
|
||||
#define mozilla_dom_SpeechSynthesis_h
|
||||
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsIObserver.h"
|
||||
#include "nsRefPtrHashtable.h"
|
||||
#include "nsString.h"
|
||||
#include "nsWeakReference.h"
|
||||
#include "nsWrapperCache.h"
|
||||
#include "js/TypeDecls.h"
|
||||
|
||||
#include "SpeechSynthesisUtterance.h"
|
||||
#include "SpeechSynthesisVoice.h"
|
||||
|
||||
class nsIDOMWindow;
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class nsSpeechTask;
|
||||
|
||||
class SpeechSynthesis final : public DOMEventTargetHelper
|
||||
, public nsIObserver
|
||||
, public nsSupportsWeakReference
|
||||
{
|
||||
public:
|
||||
explicit SpeechSynthesis(nsPIDOMWindowInner* aParent);
|
||||
|
||||
NS_DECL_ISUPPORTS_INHERITED
|
||||
NS_DECL_CYCLE_COLLECTION_CLASS_INHERITED(SpeechSynthesis, DOMEventTargetHelper)
|
||||
NS_DECL_NSIOBSERVER
|
||||
|
||||
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
|
||||
|
||||
bool Pending() const;
|
||||
|
||||
bool Speaking() const;
|
||||
|
||||
bool Paused() const;
|
||||
|
||||
bool HasEmptyQueue() const;
|
||||
|
||||
void Speak(SpeechSynthesisUtterance& aUtterance);
|
||||
|
||||
void Cancel();
|
||||
|
||||
void Pause();
|
||||
|
||||
void Resume();
|
||||
|
||||
void OnEnd(const nsSpeechTask* aTask);
|
||||
|
||||
void GetVoices(nsTArray< RefPtr<SpeechSynthesisVoice> >& aResult);
|
||||
|
||||
void ForceEnd();
|
||||
|
||||
IMPL_EVENT_HANDLER(voiceschanged)
|
||||
|
||||
private:
|
||||
virtual ~SpeechSynthesis();
|
||||
|
||||
void AdvanceQueue();
|
||||
|
||||
bool HasVoices() const;
|
||||
|
||||
nsTArray<RefPtr<SpeechSynthesisUtterance> > mSpeechQueue;
|
||||
|
||||
RefPtr<nsSpeechTask> mCurrentTask;
|
||||
|
||||
nsRefPtrHashtable<nsStringHashKey, SpeechSynthesisVoice> mVoiceCache;
|
||||
|
||||
bool mHoldQueue;
|
||||
|
||||
uint64_t mInnerID;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
#endif
|
||||
178
dom/media/webspeech/synth/SpeechSynthesisUtterance.cpp
Normal file
178
dom/media/webspeech/synth/SpeechSynthesisUtterance.cpp
Normal file
|
|
@ -0,0 +1,178 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsCycleCollectionParticipant.h"
|
||||
#include "nsGkAtoms.h"
|
||||
|
||||
#include "mozilla/dom/SpeechSynthesisEvent.h"
|
||||
#include "mozilla/dom/SpeechSynthesisUtteranceBinding.h"
|
||||
#include "SpeechSynthesisUtterance.h"
|
||||
#include "SpeechSynthesisVoice.h"
|
||||
|
||||
#include <stdlib.h>
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION_INHERITED(SpeechSynthesisUtterance,
|
||||
DOMEventTargetHelper, mVoice);
|
||||
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION_INHERITED(SpeechSynthesisUtterance)
|
||||
NS_INTERFACE_MAP_END_INHERITING(DOMEventTargetHelper)
|
||||
|
||||
NS_IMPL_ADDREF_INHERITED(SpeechSynthesisUtterance, DOMEventTargetHelper)
|
||||
NS_IMPL_RELEASE_INHERITED(SpeechSynthesisUtterance, DOMEventTargetHelper)
|
||||
|
||||
SpeechSynthesisUtterance::SpeechSynthesisUtterance(nsPIDOMWindowInner* aOwnerWindow,
|
||||
const nsAString& text)
|
||||
: DOMEventTargetHelper(aOwnerWindow)
|
||||
, mText(text)
|
||||
, mVolume(1)
|
||||
, mRate(1)
|
||||
, mPitch(1)
|
||||
, mState(STATE_NONE)
|
||||
, mPaused(false)
|
||||
{
|
||||
}
|
||||
|
||||
SpeechSynthesisUtterance::~SpeechSynthesisUtterance() {}
|
||||
|
||||
JSObject*
|
||||
SpeechSynthesisUtterance::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
|
||||
{
|
||||
return SpeechSynthesisUtteranceBinding::Wrap(aCx, this, aGivenProto);
|
||||
}
|
||||
|
||||
nsISupports*
|
||||
SpeechSynthesisUtterance::GetParentObject() const
|
||||
{
|
||||
return GetOwner();
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechSynthesisUtterance>
|
||||
SpeechSynthesisUtterance::Constructor(GlobalObject& aGlobal,
|
||||
ErrorResult& aRv)
|
||||
{
|
||||
return Constructor(aGlobal, EmptyString(), aRv);
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechSynthesisUtterance>
|
||||
SpeechSynthesisUtterance::Constructor(GlobalObject& aGlobal,
|
||||
const nsAString& aText,
|
||||
ErrorResult& aRv)
|
||||
{
|
||||
nsCOMPtr<nsPIDOMWindowInner> win = do_QueryInterface(aGlobal.GetAsSupports());
|
||||
|
||||
if (!win) {
|
||||
aRv.Throw(NS_ERROR_FAILURE);
|
||||
}
|
||||
|
||||
MOZ_ASSERT(win->IsInnerWindow());
|
||||
RefPtr<SpeechSynthesisUtterance> object =
|
||||
new SpeechSynthesisUtterance(win, aText);
|
||||
return object.forget();
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisUtterance::GetText(nsString& aResult) const
|
||||
{
|
||||
aResult = mText;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisUtterance::SetText(const nsAString& aText)
|
||||
{
|
||||
mText = aText;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisUtterance::GetLang(nsString& aResult) const
|
||||
{
|
||||
aResult = mLang;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisUtterance::SetLang(const nsAString& aLang)
|
||||
{
|
||||
mLang = aLang;
|
||||
}
|
||||
|
||||
SpeechSynthesisVoice*
|
||||
SpeechSynthesisUtterance::GetVoice() const
|
||||
{
|
||||
return mVoice;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisUtterance::SetVoice(SpeechSynthesisVoice* aVoice)
|
||||
{
|
||||
mVoice = aVoice;
|
||||
}
|
||||
|
||||
float
|
||||
SpeechSynthesisUtterance::Volume() const
|
||||
{
|
||||
return mVolume;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisUtterance::SetVolume(float aVolume)
|
||||
{
|
||||
mVolume = std::max<float>(std::min<float>(aVolume, 1), 0);
|
||||
}
|
||||
|
||||
float
|
||||
SpeechSynthesisUtterance::Rate() const
|
||||
{
|
||||
return mRate;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisUtterance::SetRate(float aRate)
|
||||
{
|
||||
mRate = std::max<float>(std::min<float>(aRate, 10), 0.1f);
|
||||
}
|
||||
|
||||
float
|
||||
SpeechSynthesisUtterance::Pitch() const
|
||||
{
|
||||
return mPitch;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisUtterance::SetPitch(float aPitch)
|
||||
{
|
||||
mPitch = std::max<float>(std::min<float>(aPitch, 2), 0);
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisUtterance::GetChosenVoiceURI(nsString& aResult) const
|
||||
{
|
||||
aResult = mChosenVoiceURI;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisUtterance::DispatchSpeechSynthesisEvent(const nsAString& aEventType,
|
||||
uint32_t aCharIndex,
|
||||
float aElapsedTime,
|
||||
const nsAString& aName)
|
||||
{
|
||||
SpeechSynthesisEventInit init;
|
||||
init.mBubbles = false;
|
||||
init.mCancelable = false;
|
||||
init.mUtterance = this;
|
||||
init.mCharIndex = aCharIndex;
|
||||
init.mElapsedTime = aElapsedTime;
|
||||
init.mName = aName;
|
||||
|
||||
RefPtr<SpeechSynthesisEvent> event =
|
||||
SpeechSynthesisEvent::Constructor(this, aEventType, init);
|
||||
DispatchTrustedEvent(event);
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
124
dom/media/webspeech/synth/SpeechSynthesisUtterance.h
Normal file
124
dom/media/webspeech/synth/SpeechSynthesisUtterance.h
Normal file
|
|
@ -0,0 +1,124 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechSynthesisUtterance_h
|
||||
#define mozilla_dom_SpeechSynthesisUtterance_h
|
||||
|
||||
#include "mozilla/DOMEventTargetHelper.h"
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsString.h"
|
||||
#include "js/TypeDecls.h"
|
||||
|
||||
#include "nsSpeechTask.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class SpeechSynthesisVoice;
|
||||
class SpeechSynthesis;
|
||||
class nsSynthVoiceRegistry;
|
||||
|
||||
class SpeechSynthesisUtterance final : public DOMEventTargetHelper
|
||||
{
|
||||
friend class SpeechSynthesis;
|
||||
friend class nsSpeechTask;
|
||||
friend class nsSynthVoiceRegistry;
|
||||
|
||||
public:
|
||||
SpeechSynthesisUtterance(nsPIDOMWindowInner* aOwnerWindow, const nsAString& aText);
|
||||
|
||||
NS_DECL_ISUPPORTS_INHERITED
|
||||
NS_DECL_CYCLE_COLLECTION_CLASS_INHERITED(SpeechSynthesisUtterance,
|
||||
DOMEventTargetHelper)
|
||||
NS_REALLY_FORWARD_NSIDOMEVENTTARGET(DOMEventTargetHelper)
|
||||
|
||||
nsISupports* GetParentObject() const;
|
||||
|
||||
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
|
||||
|
||||
static
|
||||
already_AddRefed<SpeechSynthesisUtterance> Constructor(GlobalObject& aGlobal,
|
||||
ErrorResult& aRv);
|
||||
static
|
||||
already_AddRefed<SpeechSynthesisUtterance> Constructor(GlobalObject& aGlobal,
|
||||
const nsAString& aText,
|
||||
ErrorResult& aRv);
|
||||
|
||||
void GetText(nsString& aResult) const;
|
||||
|
||||
void SetText(const nsAString& aText);
|
||||
|
||||
void GetLang(nsString& aResult) const;
|
||||
|
||||
void SetLang(const nsAString& aLang);
|
||||
|
||||
SpeechSynthesisVoice* GetVoice() const;
|
||||
|
||||
void SetVoice(SpeechSynthesisVoice* aVoice);
|
||||
|
||||
float Volume() const;
|
||||
|
||||
void SetVolume(float aVolume);
|
||||
|
||||
float Rate() const;
|
||||
|
||||
void SetRate(float aRate);
|
||||
|
||||
float Pitch() const;
|
||||
|
||||
void SetPitch(float aPitch);
|
||||
|
||||
void GetChosenVoiceURI(nsString& aResult) const;
|
||||
|
||||
enum {
|
||||
STATE_NONE,
|
||||
STATE_PENDING,
|
||||
STATE_SPEAKING,
|
||||
STATE_ENDED
|
||||
};
|
||||
|
||||
uint32_t GetState() { return mState; }
|
||||
|
||||
bool IsPaused() { return mPaused; }
|
||||
|
||||
IMPL_EVENT_HANDLER(start)
|
||||
IMPL_EVENT_HANDLER(end)
|
||||
IMPL_EVENT_HANDLER(error)
|
||||
IMPL_EVENT_HANDLER(pause)
|
||||
IMPL_EVENT_HANDLER(resume)
|
||||
IMPL_EVENT_HANDLER(mark)
|
||||
IMPL_EVENT_HANDLER(boundary)
|
||||
|
||||
private:
|
||||
virtual ~SpeechSynthesisUtterance();
|
||||
|
||||
void DispatchSpeechSynthesisEvent(const nsAString& aEventType,
|
||||
uint32_t aCharIndex,
|
||||
float aElapsedTime, const nsAString& aName);
|
||||
|
||||
nsString mText;
|
||||
|
||||
nsString mLang;
|
||||
|
||||
float mVolume;
|
||||
|
||||
float mRate;
|
||||
|
||||
float mPitch;
|
||||
|
||||
nsString mChosenVoiceURI;
|
||||
|
||||
uint32_t mState;
|
||||
|
||||
bool mPaused;
|
||||
|
||||
RefPtr<SpeechSynthesisVoice> mVoice;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
94
dom/media/webspeech/synth/SpeechSynthesisVoice.cpp
Normal file
94
dom/media/webspeech/synth/SpeechSynthesisVoice.cpp
Normal file
|
|
@ -0,0 +1,94 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "SpeechSynthesis.h"
|
||||
#include "nsSynthVoiceRegistry.h"
|
||||
#include "mozilla/dom/SpeechSynthesisVoiceBinding.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechSynthesisVoice, mParent)
|
||||
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechSynthesisVoice)
|
||||
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechSynthesisVoice)
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechSynthesisVoice)
|
||||
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
|
||||
NS_INTERFACE_MAP_ENTRY(nsISupports)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
SpeechSynthesisVoice::SpeechSynthesisVoice(nsISupports* aParent,
|
||||
const nsAString& aUri)
|
||||
: mParent(aParent)
|
||||
, mUri(aUri)
|
||||
{
|
||||
}
|
||||
|
||||
SpeechSynthesisVoice::~SpeechSynthesisVoice()
|
||||
{
|
||||
}
|
||||
|
||||
JSObject*
|
||||
SpeechSynthesisVoice::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
|
||||
{
|
||||
return SpeechSynthesisVoiceBinding::Wrap(aCx, this, aGivenProto);
|
||||
}
|
||||
|
||||
nsISupports*
|
||||
SpeechSynthesisVoice::GetParentObject() const
|
||||
{
|
||||
return mParent;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisVoice::GetVoiceURI(nsString& aRetval) const
|
||||
{
|
||||
aRetval = mUri;
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisVoice::GetName(nsString& aRetval) const
|
||||
{
|
||||
DebugOnly<nsresult> rv =
|
||||
nsSynthVoiceRegistry::GetInstance()->GetVoiceName(mUri, aRetval);
|
||||
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv),
|
||||
"Failed to get SpeechSynthesisVoice.name");
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisVoice::GetLang(nsString& aRetval) const
|
||||
{
|
||||
DebugOnly<nsresult> rv =
|
||||
nsSynthVoiceRegistry::GetInstance()->GetVoiceLang(mUri, aRetval);
|
||||
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv),
|
||||
"Failed to get SpeechSynthesisVoice.lang");
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisVoice::LocalService() const
|
||||
{
|
||||
bool isLocal;
|
||||
DebugOnly<nsresult> rv =
|
||||
nsSynthVoiceRegistry::GetInstance()->IsLocalVoice(mUri, &isLocal);
|
||||
NS_WARNING_ASSERTION(
|
||||
NS_SUCCEEDED(rv), "Failed to get SpeechSynthesisVoice.localService");
|
||||
|
||||
return isLocal;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisVoice::Default() const
|
||||
{
|
||||
bool isDefault;
|
||||
DebugOnly<nsresult> rv =
|
||||
nsSynthVoiceRegistry::GetInstance()->IsDefaultVoice(mUri, &isDefault);
|
||||
NS_WARNING_ASSERTION(
|
||||
NS_SUCCEEDED(rv), "Failed to get SpeechSynthesisVoice.default");
|
||||
|
||||
return isDefault;
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
60
dom/media/webspeech/synth/SpeechSynthesisVoice.h
Normal file
60
dom/media/webspeech/synth/SpeechSynthesisVoice.h
Normal file
|
|
@ -0,0 +1,60 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechSynthesisVoice_h
|
||||
#define mozilla_dom_SpeechSynthesisVoice_h
|
||||
|
||||
#include "nsCOMPtr.h"
|
||||
#include "nsString.h"
|
||||
#include "nsWrapperCache.h"
|
||||
#include "js/TypeDecls.h"
|
||||
|
||||
#include "nsISpeechService.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class nsSynthVoiceRegistry;
|
||||
class SpeechSynthesis;
|
||||
|
||||
class SpeechSynthesisVoice final : public nsISupports,
|
||||
public nsWrapperCache
|
||||
{
|
||||
friend class nsSynthVoiceRegistry;
|
||||
friend class SpeechSynthesis;
|
||||
|
||||
public:
|
||||
SpeechSynthesisVoice(nsISupports* aParent, const nsAString& aUri);
|
||||
|
||||
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
|
||||
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechSynthesisVoice)
|
||||
|
||||
nsISupports* GetParentObject() const;
|
||||
|
||||
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
|
||||
|
||||
void GetVoiceURI(nsString& aRetval) const;
|
||||
|
||||
void GetName(nsString& aRetval) const;
|
||||
|
||||
void GetLang(nsString& aRetval) const;
|
||||
|
||||
bool LocalService() const;
|
||||
|
||||
bool Default() const;
|
||||
|
||||
private:
|
||||
virtual ~SpeechSynthesisVoice();
|
||||
|
||||
nsCOMPtr<nsISupports> mParent;
|
||||
|
||||
nsString mUri;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
|
|
@ -0,0 +1,57 @@
|
|||
/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim: set ts=8 sts=2 et sw=2 tw=80: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "mozilla/ModuleUtils.h"
|
||||
#include "nsIClassInfoImpl.h"
|
||||
|
||||
#include "OSXSpeechSynthesizerService.h"
|
||||
|
||||
using namespace mozilla::dom;
|
||||
|
||||
#define OSXSPEECHSYNTHESIZERSERVICE_CID \
|
||||
{0x914e73b4, 0x6337, 0x4bef, {0x97, 0xf3, 0x4d, 0x06, 0x9e, 0x05, 0x3a, 0x12}}
|
||||
|
||||
#define OSXSPEECHSYNTHESIZERSERVICE_CONTRACTID "@mozilla.org/synthsystem;1"
|
||||
|
||||
// Defines OSXSpeechSynthesizerServiceConstructor
|
||||
NS_GENERIC_FACTORY_SINGLETON_CONSTRUCTOR(OSXSpeechSynthesizerService,
|
||||
OSXSpeechSynthesizerService::GetInstanceForService)
|
||||
|
||||
// Defines kOSXSERVICE_CID
|
||||
NS_DEFINE_NAMED_CID(OSXSPEECHSYNTHESIZERSERVICE_CID);
|
||||
|
||||
static const mozilla::Module::CIDEntry kCIDs[] = {
|
||||
{ &kOSXSPEECHSYNTHESIZERSERVICE_CID, true, nullptr, OSXSpeechSynthesizerServiceConstructor },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static const mozilla::Module::ContractIDEntry kContracts[] = {
|
||||
{ OSXSPEECHSYNTHESIZERSERVICE_CONTRACTID, &kOSXSPEECHSYNTHESIZERSERVICE_CID },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static const mozilla::Module::CategoryEntry kCategories[] = {
|
||||
{ "speech-synth-started", "OSX Speech Synth", OSXSPEECHSYNTHESIZERSERVICE_CONTRACTID },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static void
|
||||
UnloadOSXSpeechSynthesizerModule()
|
||||
{
|
||||
OSXSpeechSynthesizerService::Shutdown();
|
||||
}
|
||||
|
||||
static const mozilla::Module kModule = {
|
||||
mozilla::Module::kVersion,
|
||||
kCIDs,
|
||||
kContracts,
|
||||
kCategories,
|
||||
nullptr,
|
||||
nullptr,
|
||||
UnloadOSXSpeechSynthesizerModule
|
||||
};
|
||||
|
||||
NSMODULE_DEFN(osxsynth) = &kModule;
|
||||
|
|
@ -0,0 +1,44 @@
|
|||
/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim: set ts=8 sts=2 et sw=2 tw=80: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_OsxSpeechSynthesizerService_h
|
||||
#define mozilla_dom_OsxSpeechSynthesizerService_h
|
||||
|
||||
#include "nsISpeechService.h"
|
||||
#include "nsIObserver.h"
|
||||
#include "mozilla/StaticPtr.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class OSXSpeechSynthesizerService final : public nsISpeechService
|
||||
, public nsIObserver
|
||||
{
|
||||
public:
|
||||
NS_DECL_THREADSAFE_ISUPPORTS
|
||||
NS_DECL_NSISPEECHSERVICE
|
||||
NS_DECL_NSIOBSERVER
|
||||
|
||||
bool Init();
|
||||
|
||||
static OSXSpeechSynthesizerService* GetInstance();
|
||||
static already_AddRefed<OSXSpeechSynthesizerService> GetInstanceForService();
|
||||
static void Shutdown();
|
||||
|
||||
private:
|
||||
OSXSpeechSynthesizerService();
|
||||
virtual ~OSXSpeechSynthesizerService();
|
||||
|
||||
bool RegisterVoices();
|
||||
|
||||
bool mInitialized;
|
||||
static mozilla::StaticRefPtr<OSXSpeechSynthesizerService> sSingleton;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
499
dom/media/webspeech/synth/cocoa/OSXSpeechSynthesizerService.mm
Normal file
499
dom/media/webspeech/synth/cocoa/OSXSpeechSynthesizerService.mm
Normal file
|
|
@ -0,0 +1,499 @@
|
|||
/* -*- Mode: Objective-C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim: set ts=2 sw=2 et tw=80: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsISupports.h"
|
||||
#include "nsServiceManagerUtils.h"
|
||||
#include "nsObjCExceptions.h"
|
||||
#include "nsCocoaUtils.h"
|
||||
#include "nsThreadUtils.h"
|
||||
#include "mozilla/dom/nsSynthVoiceRegistry.h"
|
||||
#include "mozilla/dom/nsSpeechTask.h"
|
||||
#include "mozilla/Preferences.h"
|
||||
#include "mozilla/Assertions.h"
|
||||
#include "OSXSpeechSynthesizerService.h"
|
||||
|
||||
#import <Cocoa/Cocoa.h>
|
||||
|
||||
// We can escape the default delimiters ("[[" and "]]") by temporarily
|
||||
// changing the delimiters just before they appear, and changing them back
|
||||
// just after.
|
||||
#define DLIM_ESCAPE_START "[[dlim (( ))]]"
|
||||
#define DLIM_ESCAPE_END "((dlim [[ ]]))"
|
||||
|
||||
using namespace mozilla;
|
||||
|
||||
class SpeechTaskCallback final : public nsISpeechTaskCallback
|
||||
{
|
||||
public:
|
||||
SpeechTaskCallback(nsISpeechTask* aTask,
|
||||
NSSpeechSynthesizer* aSynth,
|
||||
const nsTArray<size_t>& aOffsets)
|
||||
: mTask(aTask)
|
||||
, mSpeechSynthesizer(aSynth)
|
||||
, mOffsets(aOffsets)
|
||||
{
|
||||
mStartingTime = TimeStamp::Now();
|
||||
}
|
||||
|
||||
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
|
||||
NS_DECL_CYCLE_COLLECTION_CLASS_AMBIGUOUS(SpeechTaskCallback, nsISpeechTaskCallback)
|
||||
|
||||
NS_DECL_NSISPEECHTASKCALLBACK
|
||||
|
||||
void OnWillSpeakWord(uint32_t aIndex);
|
||||
void OnError(uint32_t aIndex);
|
||||
void OnDidFinishSpeaking();
|
||||
|
||||
private:
|
||||
virtual ~SpeechTaskCallback()
|
||||
{
|
||||
[mSpeechSynthesizer release];
|
||||
}
|
||||
|
||||
float GetTimeDurationFromStart();
|
||||
|
||||
nsCOMPtr<nsISpeechTask> mTask;
|
||||
NSSpeechSynthesizer* mSpeechSynthesizer;
|
||||
TimeStamp mStartingTime;
|
||||
uint32_t mCurrentIndex;
|
||||
nsTArray<size_t> mOffsets;
|
||||
};
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION(SpeechTaskCallback, mTask);
|
||||
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechTaskCallback)
|
||||
NS_INTERFACE_MAP_ENTRY(nsISpeechTaskCallback)
|
||||
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsISpeechTaskCallback)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechTaskCallback)
|
||||
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechTaskCallback)
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechTaskCallback::OnCancel()
|
||||
{
|
||||
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
|
||||
|
||||
[mSpeechSynthesizer stopSpeaking];
|
||||
return NS_OK;
|
||||
|
||||
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechTaskCallback::OnPause()
|
||||
{
|
||||
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
|
||||
|
||||
[mSpeechSynthesizer pauseSpeakingAtBoundary:NSSpeechImmediateBoundary];
|
||||
if (!mTask) {
|
||||
// When calling pause() on child porcess, it may not receive end event
|
||||
// from chrome process yet.
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
mTask->DispatchPause(GetTimeDurationFromStart(), mCurrentIndex);
|
||||
return NS_OK;
|
||||
|
||||
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechTaskCallback::OnResume()
|
||||
{
|
||||
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
|
||||
|
||||
[mSpeechSynthesizer continueSpeaking];
|
||||
if (!mTask) {
|
||||
// When calling resume() on child porcess, it may not receive end event
|
||||
// from chrome process yet.
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
mTask->DispatchResume(GetTimeDurationFromStart(), mCurrentIndex);
|
||||
return NS_OK;
|
||||
|
||||
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechTaskCallback::OnVolumeChanged(float aVolume)
|
||||
{
|
||||
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
|
||||
|
||||
[mSpeechSynthesizer setObject:[NSNumber numberWithFloat:aVolume]
|
||||
forProperty:NSSpeechVolumeProperty error:nil];
|
||||
return NS_OK;
|
||||
|
||||
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
|
||||
}
|
||||
|
||||
float
|
||||
SpeechTaskCallback::GetTimeDurationFromStart()
|
||||
{
|
||||
TimeDuration duration = TimeStamp::Now() - mStartingTime;
|
||||
return duration.ToMilliseconds();
|
||||
}
|
||||
|
||||
void
|
||||
SpeechTaskCallback::OnWillSpeakWord(uint32_t aIndex)
|
||||
{
|
||||
mCurrentIndex = aIndex < mOffsets.Length() ? mOffsets[aIndex] : mCurrentIndex;
|
||||
if (!mTask) {
|
||||
return;
|
||||
}
|
||||
mTask->DispatchBoundary(NS_LITERAL_STRING("word"),
|
||||
GetTimeDurationFromStart(), mCurrentIndex);
|
||||
}
|
||||
|
||||
void
|
||||
SpeechTaskCallback::OnError(uint32_t aIndex)
|
||||
{
|
||||
if (!mTask) {
|
||||
return;
|
||||
}
|
||||
mTask->DispatchError(GetTimeDurationFromStart(), aIndex);
|
||||
}
|
||||
|
||||
void
|
||||
SpeechTaskCallback::OnDidFinishSpeaking()
|
||||
{
|
||||
mTask->DispatchEnd(GetTimeDurationFromStart(), mCurrentIndex);
|
||||
// no longer needed
|
||||
[mSpeechSynthesizer setDelegate:nil];
|
||||
mTask = nullptr;
|
||||
}
|
||||
|
||||
@interface SpeechDelegate : NSObject<NSSpeechSynthesizerDelegate>
|
||||
{
|
||||
@private
|
||||
SpeechTaskCallback* mCallback;
|
||||
}
|
||||
|
||||
- (id)initWithCallback:(SpeechTaskCallback*)aCallback;
|
||||
@end
|
||||
|
||||
@implementation SpeechDelegate
|
||||
- (id)initWithCallback:(SpeechTaskCallback*)aCallback
|
||||
{
|
||||
[super init];
|
||||
mCallback = aCallback;
|
||||
return self;
|
||||
}
|
||||
|
||||
- (void)speechSynthesizer:(NSSpeechSynthesizer *)aSender
|
||||
willSpeakWord:(NSRange)aRange ofString:(NSString*)aString
|
||||
{
|
||||
mCallback->OnWillSpeakWord(aRange.location);
|
||||
}
|
||||
|
||||
- (void)speechSynthesizer:(NSSpeechSynthesizer *)aSender
|
||||
didFinishSpeaking:(BOOL)aFinishedSpeaking
|
||||
{
|
||||
mCallback->OnDidFinishSpeaking();
|
||||
}
|
||||
|
||||
- (void)speechSynthesizer:(NSSpeechSynthesizer*)aSender
|
||||
didEncounterErrorAtIndex:(NSUInteger)aCharacterIndex
|
||||
ofString:(NSString*)aString
|
||||
message:(NSString*)aMessage
|
||||
{
|
||||
mCallback->OnError(aCharacterIndex);
|
||||
}
|
||||
@end
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
struct OSXVoice
|
||||
{
|
||||
OSXVoice() : mIsDefault(false)
|
||||
{
|
||||
}
|
||||
|
||||
nsString mUri;
|
||||
nsString mName;
|
||||
nsString mLocale;
|
||||
bool mIsDefault;
|
||||
};
|
||||
|
||||
class RegisterVoicesRunnable final : public Runnable
|
||||
{
|
||||
public:
|
||||
RegisterVoicesRunnable(OSXSpeechSynthesizerService* aSpeechService,
|
||||
nsTArray<OSXVoice>& aList)
|
||||
: mSpeechService(aSpeechService)
|
||||
, mVoices(aList)
|
||||
{
|
||||
}
|
||||
|
||||
NS_IMETHOD Run() override;
|
||||
|
||||
private:
|
||||
~RegisterVoicesRunnable()
|
||||
{
|
||||
}
|
||||
|
||||
// This runnable always use sync mode. It is unnecesarry to reference object
|
||||
OSXSpeechSynthesizerService* mSpeechService;
|
||||
nsTArray<OSXVoice>& mVoices;
|
||||
};
|
||||
|
||||
NS_IMETHODIMP
|
||||
RegisterVoicesRunnable::Run()
|
||||
{
|
||||
nsresult rv;
|
||||
nsCOMPtr<nsISynthVoiceRegistry> registry =
|
||||
do_GetService(NS_SYNTHVOICEREGISTRY_CONTRACTID, &rv);
|
||||
if (!registry) {
|
||||
return rv;
|
||||
}
|
||||
|
||||
for (OSXVoice voice : mVoices) {
|
||||
rv = registry->AddVoice(mSpeechService, voice.mUri, voice.mName, voice.mLocale, true, false);
|
||||
if (NS_WARN_IF(NS_FAILED(rv))) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (voice.mIsDefault) {
|
||||
registry->SetDefaultVoice(voice.mUri, true);
|
||||
}
|
||||
}
|
||||
|
||||
registry->NotifyVoicesChanged();
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
class EnumVoicesRunnable final : public Runnable
|
||||
{
|
||||
public:
|
||||
explicit EnumVoicesRunnable(OSXSpeechSynthesizerService* aSpeechService)
|
||||
: mSpeechService(aSpeechService)
|
||||
{
|
||||
}
|
||||
|
||||
NS_IMETHOD Run() override;
|
||||
|
||||
private:
|
||||
~EnumVoicesRunnable()
|
||||
{
|
||||
}
|
||||
|
||||
RefPtr<OSXSpeechSynthesizerService> mSpeechService;
|
||||
};
|
||||
|
||||
NS_IMETHODIMP
|
||||
EnumVoicesRunnable::Run()
|
||||
{
|
||||
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
|
||||
|
||||
AutoTArray<OSXVoice, 64> list;
|
||||
|
||||
NSArray* voices = [NSSpeechSynthesizer availableVoices];
|
||||
NSString* defaultVoice = [NSSpeechSynthesizer defaultVoice];
|
||||
|
||||
for (NSString* voice in voices) {
|
||||
OSXVoice item;
|
||||
|
||||
NSDictionary* attr = [NSSpeechSynthesizer attributesForVoice:voice];
|
||||
|
||||
nsAutoString identifier;
|
||||
nsCocoaUtils::GetStringForNSString([attr objectForKey:NSVoiceIdentifier],
|
||||
identifier);
|
||||
|
||||
nsCocoaUtils::GetStringForNSString([attr objectForKey:NSVoiceName], item.mName);
|
||||
|
||||
nsCocoaUtils::GetStringForNSString(
|
||||
[attr objectForKey:NSVoiceLocaleIdentifier], item.mLocale);
|
||||
item.mLocale.ReplaceChar('_', '-');
|
||||
|
||||
item.mUri.AssignLiteral("urn:moz-tts:osx:");
|
||||
item.mUri.Append(identifier);
|
||||
|
||||
if ([voice isEqualToString:defaultVoice]) {
|
||||
item.mIsDefault = true;
|
||||
}
|
||||
|
||||
list.AppendElement(item);
|
||||
}
|
||||
|
||||
RefPtr<RegisterVoicesRunnable> runnable = new RegisterVoicesRunnable(mSpeechService, list);
|
||||
NS_DispatchToMainThread(runnable, NS_DISPATCH_SYNC);
|
||||
|
||||
return NS_OK;
|
||||
|
||||
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
|
||||
}
|
||||
|
||||
StaticRefPtr<OSXSpeechSynthesizerService> OSXSpeechSynthesizerService::sSingleton;
|
||||
|
||||
NS_INTERFACE_MAP_BEGIN(OSXSpeechSynthesizerService)
|
||||
NS_INTERFACE_MAP_ENTRY(nsISpeechService)
|
||||
NS_INTERFACE_MAP_ENTRY(nsIObserver)
|
||||
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsISpeechService)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
NS_IMPL_ADDREF(OSXSpeechSynthesizerService)
|
||||
NS_IMPL_RELEASE(OSXSpeechSynthesizerService)
|
||||
|
||||
OSXSpeechSynthesizerService::OSXSpeechSynthesizerService()
|
||||
: mInitialized(false)
|
||||
{
|
||||
}
|
||||
|
||||
OSXSpeechSynthesizerService::~OSXSpeechSynthesizerService()
|
||||
{
|
||||
}
|
||||
|
||||
bool
|
||||
OSXSpeechSynthesizerService::Init()
|
||||
{
|
||||
if (Preferences::GetBool("media.webspeech.synth.test") ||
|
||||
!Preferences::GetBool("media.webspeech.synth.enabled")) {
|
||||
// When test is enabled, we shouldn't add OS backend (Bug 1160844)
|
||||
return false;
|
||||
}
|
||||
|
||||
nsCOMPtr<nsIThread> thread;
|
||||
if (NS_FAILED(NS_NewNamedThread("SpeechWorker", getter_AddRefs(thread)))) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Get all the voices and register in the SynthVoiceRegistry
|
||||
nsCOMPtr<nsIRunnable> runnable = new EnumVoicesRunnable(this);
|
||||
thread->Dispatch(runnable, NS_DISPATCH_NORMAL);
|
||||
|
||||
mInitialized = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
OSXSpeechSynthesizerService::Speak(const nsAString& aText,
|
||||
const nsAString& aUri,
|
||||
float aVolume,
|
||||
float aRate,
|
||||
float aPitch,
|
||||
nsISpeechTask* aTask)
|
||||
{
|
||||
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
|
||||
|
||||
MOZ_ASSERT(StringBeginsWith(aUri, NS_LITERAL_STRING("urn:moz-tts:osx:")),
|
||||
"OSXSpeechSynthesizerService doesn't allow this voice URI");
|
||||
|
||||
NSSpeechSynthesizer* synth = [[NSSpeechSynthesizer alloc] init];
|
||||
// strlen("urn:moz-tts:osx:") == 16
|
||||
NSString* identifier = nsCocoaUtils::ToNSString(Substring(aUri, 16));
|
||||
[synth setVoice:identifier];
|
||||
|
||||
// default rate is 180-220
|
||||
[synth setObject:[NSNumber numberWithInt:aRate * 200]
|
||||
forProperty:NSSpeechRateProperty error:nil];
|
||||
// volume allows 0.0-1.0
|
||||
[synth setObject:[NSNumber numberWithFloat:aVolume]
|
||||
forProperty:NSSpeechVolumeProperty error:nil];
|
||||
// Use default pitch value to calculate this
|
||||
NSNumber* defaultPitch =
|
||||
[synth objectForProperty:NSSpeechPitchBaseProperty error:nil];
|
||||
if (defaultPitch) {
|
||||
int newPitch = [defaultPitch intValue] * (aPitch / 2 + 0.5);
|
||||
[synth setObject:[NSNumber numberWithInt:newPitch]
|
||||
forProperty:NSSpeechPitchBaseProperty error:nil];
|
||||
}
|
||||
|
||||
nsAutoString escapedText;
|
||||
// We need to map the the offsets from the given text to the escaped text.
|
||||
// The index of the offsets array is the position in the escaped text,
|
||||
// the element value is the position in the user-supplied text.
|
||||
nsTArray<size_t> offsets;
|
||||
offsets.SetCapacity(aText.Length());
|
||||
|
||||
// This loop looks for occurances of "[[" or "]]", escapes them, and
|
||||
// populates the offsets array to supply a map to the original offsets.
|
||||
for (size_t i = 0; i < aText.Length(); i++) {
|
||||
if (aText.Length() > i + 1 &&
|
||||
((aText[i] == ']' && aText[i+1] == ']') ||
|
||||
(aText[i] == '[' && aText[i+1] == '['))) {
|
||||
escapedText.AppendLiteral(DLIM_ESCAPE_START);
|
||||
offsets.AppendElements(strlen(DLIM_ESCAPE_START));
|
||||
escapedText.Append(aText[i]);
|
||||
offsets.AppendElement(i);
|
||||
escapedText.Append(aText[++i]);
|
||||
offsets.AppendElement(i);
|
||||
escapedText.AppendLiteral(DLIM_ESCAPE_END);
|
||||
offsets.AppendElements(strlen(DLIM_ESCAPE_END));
|
||||
} else {
|
||||
escapedText.Append(aText[i]);
|
||||
offsets.AppendElement(i);
|
||||
}
|
||||
}
|
||||
|
||||
RefPtr<SpeechTaskCallback> callback = new SpeechTaskCallback(aTask, synth, offsets);
|
||||
nsresult rv = aTask->Setup(callback, 0, 0, 0);
|
||||
NS_ENSURE_SUCCESS(rv, rv);
|
||||
|
||||
SpeechDelegate* delegate = [[SpeechDelegate alloc] initWithCallback:callback];
|
||||
[synth setDelegate:delegate];
|
||||
[delegate release ];
|
||||
|
||||
NSString* text = nsCocoaUtils::ToNSString(escapedText);
|
||||
BOOL success = [synth startSpeakingString:text];
|
||||
NS_ENSURE_TRUE(success, NS_ERROR_FAILURE);
|
||||
|
||||
aTask->DispatchStart();
|
||||
return NS_OK;
|
||||
|
||||
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
OSXSpeechSynthesizerService::GetServiceType(SpeechServiceType* aServiceType)
|
||||
{
|
||||
*aServiceType = nsISpeechService::SERVICETYPE_INDIRECT_AUDIO;
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
OSXSpeechSynthesizerService::Observe(nsISupports* aSubject, const char* aTopic,
|
||||
const char16_t* aData)
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
OSXSpeechSynthesizerService*
|
||||
OSXSpeechSynthesizerService::GetInstance()
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
if (XRE_GetProcessType() != GeckoProcessType_Default) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
if (!sSingleton) {
|
||||
RefPtr<OSXSpeechSynthesizerService> speechService =
|
||||
new OSXSpeechSynthesizerService();
|
||||
if (speechService->Init()) {
|
||||
sSingleton = speechService;
|
||||
}
|
||||
}
|
||||
return sSingleton;
|
||||
}
|
||||
|
||||
already_AddRefed<OSXSpeechSynthesizerService>
|
||||
OSXSpeechSynthesizerService::GetInstanceForService()
|
||||
{
|
||||
RefPtr<OSXSpeechSynthesizerService> speechService = GetInstance();
|
||||
return speechService.forget();
|
||||
}
|
||||
|
||||
void
|
||||
OSXSpeechSynthesizerService::Shutdown()
|
||||
{
|
||||
if (!sSingleton) {
|
||||
return;
|
||||
}
|
||||
sSingleton = nullptr;
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
12
dom/media/webspeech/synth/cocoa/moz.build
Normal file
12
dom/media/webspeech/synth/cocoa/moz.build
Normal file
|
|
@ -0,0 +1,12 @@
|
|||
# -*- Mode: python; indent-tabs-mode: nil; tab-width: 40 -*-
|
||||
# vim: set filetype=python:
|
||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
SOURCES += [
|
||||
'OSXSpeechSynthesizerModule.cpp',
|
||||
'OSXSpeechSynthesizerService.mm'
|
||||
]
|
||||
|
||||
FINAL_LIBRARY = 'xul'
|
||||
32
dom/media/webspeech/synth/crashtests/1230428.html
Normal file
32
dom/media/webspeech/synth/crashtests/1230428.html
Normal file
|
|
@ -0,0 +1,32 @@
|
|||
<!DOCTYPE html>
|
||||
<html class="reftest-wait">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<script type="application/javascript">
|
||||
function f()
|
||||
{
|
||||
if (speechSynthesis.getVoices().length == 0) {
|
||||
// No synthesis backend to test this
|
||||
document.documentElement.removeAttribute('class');
|
||||
return;
|
||||
}
|
||||
|
||||
var s = new SpeechSynthesisUtterance("hello world");
|
||||
s.onerror = () => {
|
||||
// No synthesis backend to test this
|
||||
document.documentElement.removeAttribute('class');
|
||||
return;
|
||||
}
|
||||
s.onend = () => {
|
||||
document.documentElement.removeAttribute('class');
|
||||
};
|
||||
speechSynthesis.speak(s);
|
||||
speechSynthesis.cancel();
|
||||
speechSynthesis.pause();
|
||||
speechSynthesis.resume();
|
||||
}
|
||||
</script>
|
||||
</head>
|
||||
<body onload="f();">
|
||||
</body>
|
||||
</html>
|
||||
1
dom/media/webspeech/synth/crashtests/crashtests.list
Normal file
1
dom/media/webspeech/synth/crashtests/crashtests.list
Normal file
|
|
@ -0,0 +1 @@
|
|||
skip-if(!cocoaWidget) pref(media.webspeech.synth.enabled,true) load 1230428.html # bug 1230428
|
||||
49
dom/media/webspeech/synth/ipc/PSpeechSynthesis.ipdl
Normal file
49
dom/media/webspeech/synth/ipc/PSpeechSynthesis.ipdl
Normal file
|
|
@ -0,0 +1,49 @@
|
|||
/* -*- Mode: c++; c-basic-offset: 2; indent-tabs-mode: nil; tab-width: 40 -*- */
|
||||
/* vim: set ts=2 et sw=2 tw=80: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
|
||||
* You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
include protocol PContent;
|
||||
include protocol PSpeechSynthesisRequest;
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
struct RemoteVoice {
|
||||
nsString voiceURI;
|
||||
nsString name;
|
||||
nsString lang;
|
||||
bool localService;
|
||||
bool queued;
|
||||
};
|
||||
|
||||
sync protocol PSpeechSynthesis
|
||||
{
|
||||
manager PContent;
|
||||
manages PSpeechSynthesisRequest;
|
||||
|
||||
child:
|
||||
|
||||
async VoiceAdded(RemoteVoice aVoice);
|
||||
|
||||
async VoiceRemoved(nsString aUri);
|
||||
|
||||
async SetDefaultVoice(nsString aUri, bool aIsDefault);
|
||||
|
||||
async IsSpeakingChanged(bool aIsSpeaking);
|
||||
|
||||
async NotifyVoicesChanged();
|
||||
|
||||
parent:
|
||||
async __delete__();
|
||||
|
||||
async PSpeechSynthesisRequest(nsString aText, nsString aUri, nsString aLang,
|
||||
float aVolume, float aRate, float aPitch);
|
||||
|
||||
sync ReadVoicesAndState() returns (RemoteVoice[] aVoices,
|
||||
nsString[] aDefaults, bool aIsSpeaking);
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
46
dom/media/webspeech/synth/ipc/PSpeechSynthesisRequest.ipdl
Normal file
46
dom/media/webspeech/synth/ipc/PSpeechSynthesisRequest.ipdl
Normal file
|
|
@ -0,0 +1,46 @@
|
|||
/* -*- Mode: c++; c-basic-offset: 2; indent-tabs-mode: nil; tab-width: 40 -*- */
|
||||
/* vim: set ts=2 et sw=2 tw=80: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
|
||||
* You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
include protocol PSpeechSynthesis;
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
async protocol PSpeechSynthesisRequest
|
||||
{
|
||||
manager PSpeechSynthesis;
|
||||
|
||||
parent:
|
||||
|
||||
async __delete__();
|
||||
|
||||
async Pause();
|
||||
|
||||
async Resume();
|
||||
|
||||
async Cancel();
|
||||
|
||||
async ForceEnd();
|
||||
|
||||
async SetAudioOutputVolume(float aVolume);
|
||||
|
||||
child:
|
||||
|
||||
async OnEnd(bool aIsError, float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
async OnStart(nsString aUri);
|
||||
|
||||
async OnPause(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
async OnResume(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
async OnBoundary(nsString aName, float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
async OnMark(nsString aName, float aElapsedTime, uint32_t aCharIndex);
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
213
dom/media/webspeech/synth/ipc/SpeechSynthesisChild.cpp
Normal file
213
dom/media/webspeech/synth/ipc/SpeechSynthesisChild.cpp
Normal file
|
|
@ -0,0 +1,213 @@
|
|||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
|
||||
* You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "SpeechSynthesisChild.h"
|
||||
#include "nsSynthVoiceRegistry.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
SpeechSynthesisChild::SpeechSynthesisChild()
|
||||
{
|
||||
MOZ_COUNT_CTOR(SpeechSynthesisChild);
|
||||
}
|
||||
|
||||
SpeechSynthesisChild::~SpeechSynthesisChild()
|
||||
{
|
||||
MOZ_COUNT_DTOR(SpeechSynthesisChild);
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisChild::RecvVoiceAdded(const RemoteVoice& aVoice)
|
||||
{
|
||||
nsSynthVoiceRegistry::RecvAddVoice(aVoice);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisChild::RecvVoiceRemoved(const nsString& aUri)
|
||||
{
|
||||
nsSynthVoiceRegistry::RecvRemoveVoice(aUri);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisChild::RecvSetDefaultVoice(const nsString& aUri,
|
||||
const bool& aIsDefault)
|
||||
{
|
||||
nsSynthVoiceRegistry::RecvSetDefaultVoice(aUri, aIsDefault);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisChild::RecvIsSpeakingChanged(const bool& aIsSpeaking)
|
||||
{
|
||||
nsSynthVoiceRegistry::RecvIsSpeakingChanged(aIsSpeaking);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisChild::RecvNotifyVoicesChanged()
|
||||
{
|
||||
nsSynthVoiceRegistry::RecvNotifyVoicesChanged();
|
||||
return true;
|
||||
}
|
||||
|
||||
PSpeechSynthesisRequestChild*
|
||||
SpeechSynthesisChild::AllocPSpeechSynthesisRequestChild(const nsString& aText,
|
||||
const nsString& aLang,
|
||||
const nsString& aUri,
|
||||
const float& aVolume,
|
||||
const float& aRate,
|
||||
const float& aPitch)
|
||||
{
|
||||
MOZ_CRASH("Caller is supposed to manually construct a request!");
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisChild::DeallocPSpeechSynthesisRequestChild(PSpeechSynthesisRequestChild* aActor)
|
||||
{
|
||||
delete aActor;
|
||||
return true;
|
||||
}
|
||||
|
||||
// SpeechSynthesisRequestChild
|
||||
|
||||
SpeechSynthesisRequestChild::SpeechSynthesisRequestChild(SpeechTaskChild* aTask)
|
||||
: mTask(aTask)
|
||||
{
|
||||
mTask->mActor = this;
|
||||
MOZ_COUNT_CTOR(SpeechSynthesisRequestChild);
|
||||
}
|
||||
|
||||
SpeechSynthesisRequestChild::~SpeechSynthesisRequestChild()
|
||||
{
|
||||
MOZ_COUNT_DTOR(SpeechSynthesisRequestChild);
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestChild::RecvOnStart(const nsString& aUri)
|
||||
{
|
||||
mTask->DispatchStartImpl(aUri);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestChild::RecvOnEnd(const bool& aIsError,
|
||||
const float& aElapsedTime,
|
||||
const uint32_t& aCharIndex)
|
||||
{
|
||||
SpeechSynthesisRequestChild* actor = mTask->mActor;
|
||||
mTask->mActor = nullptr;
|
||||
|
||||
if (aIsError) {
|
||||
mTask->DispatchErrorImpl(aElapsedTime, aCharIndex);
|
||||
} else {
|
||||
mTask->DispatchEndImpl(aElapsedTime, aCharIndex);
|
||||
}
|
||||
|
||||
actor->Send__delete__(actor);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestChild::RecvOnPause(const float& aElapsedTime,
|
||||
const uint32_t& aCharIndex)
|
||||
{
|
||||
mTask->DispatchPauseImpl(aElapsedTime, aCharIndex);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestChild::RecvOnResume(const float& aElapsedTime,
|
||||
const uint32_t& aCharIndex)
|
||||
{
|
||||
mTask->DispatchResumeImpl(aElapsedTime, aCharIndex);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestChild::RecvOnBoundary(const nsString& aName,
|
||||
const float& aElapsedTime,
|
||||
const uint32_t& aCharIndex)
|
||||
{
|
||||
mTask->DispatchBoundaryImpl(aName, aElapsedTime, aCharIndex);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestChild::RecvOnMark(const nsString& aName,
|
||||
const float& aElapsedTime,
|
||||
const uint32_t& aCharIndex)
|
||||
{
|
||||
mTask->DispatchMarkImpl(aName, aElapsedTime, aCharIndex);
|
||||
return true;
|
||||
}
|
||||
|
||||
// SpeechTaskChild
|
||||
|
||||
SpeechTaskChild::SpeechTaskChild(SpeechSynthesisUtterance* aUtterance)
|
||||
: nsSpeechTask(aUtterance)
|
||||
{
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechTaskChild::Setup(nsISpeechTaskCallback* aCallback,
|
||||
uint32_t aChannels, uint32_t aRate, uint8_t argc)
|
||||
{
|
||||
MOZ_CRASH("Should never be called from child");
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechTaskChild::SendAudio(JS::Handle<JS::Value> aData, JS::Handle<JS::Value> aLandmarks,
|
||||
JSContext* aCx)
|
||||
{
|
||||
MOZ_CRASH("Should never be called from child");
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechTaskChild::SendAudioNative(int16_t* aData, uint32_t aDataLen)
|
||||
{
|
||||
MOZ_CRASH("Should never be called from child");
|
||||
}
|
||||
|
||||
void
|
||||
SpeechTaskChild::Pause()
|
||||
{
|
||||
MOZ_ASSERT(mActor);
|
||||
mActor->SendPause();
|
||||
}
|
||||
|
||||
void
|
||||
SpeechTaskChild::Resume()
|
||||
{
|
||||
MOZ_ASSERT(mActor);
|
||||
mActor->SendResume();
|
||||
}
|
||||
|
||||
void
|
||||
SpeechTaskChild::Cancel()
|
||||
{
|
||||
MOZ_ASSERT(mActor);
|
||||
mActor->SendCancel();
|
||||
}
|
||||
|
||||
void
|
||||
SpeechTaskChild::ForceEnd()
|
||||
{
|
||||
MOZ_ASSERT(mActor);
|
||||
mActor->SendForceEnd();
|
||||
}
|
||||
|
||||
void
|
||||
SpeechTaskChild::SetAudioOutputVolume(float aVolume)
|
||||
{
|
||||
if (mActor) {
|
||||
mActor->SendSetAudioOutputVolume(aVolume);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
106
dom/media/webspeech/synth/ipc/SpeechSynthesisChild.h
Normal file
106
dom/media/webspeech/synth/ipc/SpeechSynthesisChild.h
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
|
||||
* You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechSynthesisChild_h
|
||||
#define mozilla_dom_SpeechSynthesisChild_h
|
||||
|
||||
#include "mozilla/Attributes.h"
|
||||
#include "mozilla/dom/PSpeechSynthesisChild.h"
|
||||
#include "mozilla/dom/PSpeechSynthesisRequestChild.h"
|
||||
#include "nsSpeechTask.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class nsSynthVoiceRegistry;
|
||||
class SpeechSynthesisRequestChild;
|
||||
class SpeechTaskChild;
|
||||
|
||||
class SpeechSynthesisChild : public PSpeechSynthesisChild
|
||||
{
|
||||
friend class nsSynthVoiceRegistry;
|
||||
|
||||
public:
|
||||
bool RecvVoiceAdded(const RemoteVoice& aVoice) override;
|
||||
|
||||
bool RecvVoiceRemoved(const nsString& aUri) override;
|
||||
|
||||
bool RecvSetDefaultVoice(const nsString& aUri, const bool& aIsDefault) override;
|
||||
|
||||
bool RecvIsSpeakingChanged(const bool& aIsSpeaking) override;
|
||||
|
||||
bool RecvNotifyVoicesChanged() override;
|
||||
|
||||
protected:
|
||||
SpeechSynthesisChild();
|
||||
virtual ~SpeechSynthesisChild();
|
||||
|
||||
PSpeechSynthesisRequestChild* AllocPSpeechSynthesisRequestChild(const nsString& aLang,
|
||||
const nsString& aUri,
|
||||
const nsString& aText,
|
||||
const float& aVolume,
|
||||
const float& aPitch,
|
||||
const float& aRate) override;
|
||||
bool DeallocPSpeechSynthesisRequestChild(PSpeechSynthesisRequestChild* aActor) override;
|
||||
};
|
||||
|
||||
class SpeechSynthesisRequestChild : public PSpeechSynthesisRequestChild
|
||||
{
|
||||
public:
|
||||
explicit SpeechSynthesisRequestChild(SpeechTaskChild* aTask);
|
||||
virtual ~SpeechSynthesisRequestChild();
|
||||
|
||||
protected:
|
||||
bool RecvOnStart(const nsString& aUri) override;
|
||||
|
||||
bool RecvOnEnd(const bool& aIsError,
|
||||
const float& aElapsedTime,
|
||||
const uint32_t& aCharIndex) override;
|
||||
|
||||
bool RecvOnPause(const float& aElapsedTime, const uint32_t& aCharIndex) override;
|
||||
|
||||
bool RecvOnResume(const float& aElapsedTime, const uint32_t& aCharIndex) override;
|
||||
|
||||
bool RecvOnBoundary(const nsString& aName, const float& aElapsedTime,
|
||||
const uint32_t& aCharIndex) override;
|
||||
|
||||
bool RecvOnMark(const nsString& aName, const float& aElapsedTime,
|
||||
const uint32_t& aCharIndex) override;
|
||||
|
||||
RefPtr<SpeechTaskChild> mTask;
|
||||
};
|
||||
|
||||
class SpeechTaskChild : public nsSpeechTask
|
||||
{
|
||||
friend class SpeechSynthesisRequestChild;
|
||||
public:
|
||||
|
||||
explicit SpeechTaskChild(SpeechSynthesisUtterance* aUtterance);
|
||||
|
||||
NS_IMETHOD Setup(nsISpeechTaskCallback* aCallback,
|
||||
uint32_t aChannels, uint32_t aRate, uint8_t argc) override;
|
||||
|
||||
NS_IMETHOD SendAudio(JS::Handle<JS::Value> aData, JS::Handle<JS::Value> aLandmarks,
|
||||
JSContext* aCx) override;
|
||||
|
||||
NS_IMETHOD SendAudioNative(int16_t* aData, uint32_t aDataLen) override;
|
||||
|
||||
void Pause() override;
|
||||
|
||||
void Resume() override;
|
||||
|
||||
void Cancel() override;
|
||||
|
||||
void ForceEnd() override;
|
||||
|
||||
void SetAudioOutputVolume(float aVolume) override;
|
||||
|
||||
private:
|
||||
SpeechSynthesisRequestChild* mActor;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
234
dom/media/webspeech/synth/ipc/SpeechSynthesisParent.cpp
Normal file
234
dom/media/webspeech/synth/ipc/SpeechSynthesisParent.cpp
Normal file
|
|
@ -0,0 +1,234 @@
|
|||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
|
||||
* You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "SpeechSynthesisParent.h"
|
||||
#include "nsSynthVoiceRegistry.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
SpeechSynthesisParent::SpeechSynthesisParent()
|
||||
{
|
||||
MOZ_COUNT_CTOR(SpeechSynthesisParent);
|
||||
}
|
||||
|
||||
SpeechSynthesisParent::~SpeechSynthesisParent()
|
||||
{
|
||||
MOZ_COUNT_DTOR(SpeechSynthesisParent);
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisParent::ActorDestroy(ActorDestroyReason aWhy)
|
||||
{
|
||||
// Implement me! Bug 1005141
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisParent::RecvReadVoicesAndState(InfallibleTArray<RemoteVoice>* aVoices,
|
||||
InfallibleTArray<nsString>* aDefaults,
|
||||
bool* aIsSpeaking)
|
||||
{
|
||||
nsSynthVoiceRegistry::GetInstance()->SendVoicesAndState(aVoices, aDefaults,
|
||||
aIsSpeaking);
|
||||
return true;
|
||||
}
|
||||
|
||||
PSpeechSynthesisRequestParent*
|
||||
SpeechSynthesisParent::AllocPSpeechSynthesisRequestParent(const nsString& aText,
|
||||
const nsString& aLang,
|
||||
const nsString& aUri,
|
||||
const float& aVolume,
|
||||
const float& aRate,
|
||||
const float& aPitch)
|
||||
{
|
||||
RefPtr<SpeechTaskParent> task = new SpeechTaskParent(aVolume, aText);
|
||||
SpeechSynthesisRequestParent* actor = new SpeechSynthesisRequestParent(task);
|
||||
return actor;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisParent::DeallocPSpeechSynthesisRequestParent(PSpeechSynthesisRequestParent* aActor)
|
||||
{
|
||||
delete aActor;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisParent::RecvPSpeechSynthesisRequestConstructor(PSpeechSynthesisRequestParent* aActor,
|
||||
const nsString& aText,
|
||||
const nsString& aLang,
|
||||
const nsString& aUri,
|
||||
const float& aVolume,
|
||||
const float& aRate,
|
||||
const float& aPitch)
|
||||
{
|
||||
MOZ_ASSERT(aActor);
|
||||
SpeechSynthesisRequestParent* actor =
|
||||
static_cast<SpeechSynthesisRequestParent*>(aActor);
|
||||
nsSynthVoiceRegistry::GetInstance()->Speak(aText, aLang, aUri, aVolume, aRate,
|
||||
aPitch, actor->mTask);
|
||||
return true;
|
||||
}
|
||||
|
||||
// SpeechSynthesisRequestParent
|
||||
|
||||
SpeechSynthesisRequestParent::SpeechSynthesisRequestParent(SpeechTaskParent* aTask)
|
||||
: mTask(aTask)
|
||||
{
|
||||
mTask->mActor = this;
|
||||
MOZ_COUNT_CTOR(SpeechSynthesisRequestParent);
|
||||
}
|
||||
|
||||
SpeechSynthesisRequestParent::~SpeechSynthesisRequestParent()
|
||||
{
|
||||
if (mTask) {
|
||||
mTask->mActor = nullptr;
|
||||
// If we still have a task, cancel it.
|
||||
mTask->Cancel();
|
||||
}
|
||||
MOZ_COUNT_DTOR(SpeechSynthesisRequestParent);
|
||||
}
|
||||
|
||||
void
|
||||
SpeechSynthesisRequestParent::ActorDestroy(ActorDestroyReason aWhy)
|
||||
{
|
||||
// Implement me! Bug 1005141
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestParent::RecvPause()
|
||||
{
|
||||
MOZ_ASSERT(mTask);
|
||||
mTask->Pause();
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestParent::Recv__delete__()
|
||||
{
|
||||
MOZ_ASSERT(mTask);
|
||||
mTask->mActor = nullptr;
|
||||
mTask = nullptr;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestParent::RecvResume()
|
||||
{
|
||||
MOZ_ASSERT(mTask);
|
||||
mTask->Resume();
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestParent::RecvCancel()
|
||||
{
|
||||
MOZ_ASSERT(mTask);
|
||||
mTask->Cancel();
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestParent::RecvForceEnd()
|
||||
{
|
||||
MOZ_ASSERT(mTask);
|
||||
mTask->ForceEnd();
|
||||
return true;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechSynthesisRequestParent::RecvSetAudioOutputVolume(const float& aVolume)
|
||||
{
|
||||
MOZ_ASSERT(mTask);
|
||||
mTask->SetAudioOutputVolume(aVolume);
|
||||
return true;
|
||||
}
|
||||
|
||||
// SpeechTaskParent
|
||||
|
||||
nsresult
|
||||
SpeechTaskParent::DispatchStartImpl(const nsAString& aUri)
|
||||
{
|
||||
MOZ_ASSERT(mActor);
|
||||
if(NS_WARN_IF(!(mActor->SendOnStart(nsString(aUri))))) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
nsresult
|
||||
SpeechTaskParent::DispatchEndImpl(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
if (!mActor) {
|
||||
// Child is already gone.
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
if(NS_WARN_IF(!(mActor->SendOnEnd(false, aElapsedTime, aCharIndex)))) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
nsresult
|
||||
SpeechTaskParent::DispatchPauseImpl(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
MOZ_ASSERT(mActor);
|
||||
if(NS_WARN_IF(!(mActor->SendOnPause(aElapsedTime, aCharIndex)))) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
nsresult
|
||||
SpeechTaskParent::DispatchResumeImpl(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
MOZ_ASSERT(mActor);
|
||||
if(NS_WARN_IF(!(mActor->SendOnResume(aElapsedTime, aCharIndex)))) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
nsresult
|
||||
SpeechTaskParent::DispatchErrorImpl(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
MOZ_ASSERT(mActor);
|
||||
if(NS_WARN_IF(!(mActor->SendOnEnd(true, aElapsedTime, aCharIndex)))) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
nsresult
|
||||
SpeechTaskParent::DispatchBoundaryImpl(const nsAString& aName,
|
||||
float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
MOZ_ASSERT(mActor);
|
||||
if(NS_WARN_IF(!(mActor->SendOnBoundary(nsString(aName), aElapsedTime, aCharIndex)))) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
nsresult
|
||||
SpeechTaskParent::DispatchMarkImpl(const nsAString& aName,
|
||||
float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
MOZ_ASSERT(mActor);
|
||||
if(NS_WARN_IF(!(mActor->SendOnMark(nsString(aName), aElapsedTime, aCharIndex)))) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
108
dom/media/webspeech/synth/ipc/SpeechSynthesisParent.h
Normal file
108
dom/media/webspeech/synth/ipc/SpeechSynthesisParent.h
Normal file
|
|
@ -0,0 +1,108 @@
|
|||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
|
||||
* You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechSynthesisParent_h
|
||||
#define mozilla_dom_SpeechSynthesisParent_h
|
||||
|
||||
#include "mozilla/dom/PSpeechSynthesisParent.h"
|
||||
#include "mozilla/dom/PSpeechSynthesisRequestParent.h"
|
||||
#include "nsSpeechTask.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class ContentParent;
|
||||
class SpeechTaskParent;
|
||||
class SpeechSynthesisRequestParent;
|
||||
|
||||
class SpeechSynthesisParent : public PSpeechSynthesisParent
|
||||
{
|
||||
friend class ContentParent;
|
||||
friend class SpeechSynthesisRequestParent;
|
||||
|
||||
public:
|
||||
void ActorDestroy(ActorDestroyReason aWhy) override;
|
||||
|
||||
bool RecvReadVoicesAndState(InfallibleTArray<RemoteVoice>* aVoices,
|
||||
InfallibleTArray<nsString>* aDefaults,
|
||||
bool* aIsSpeaking) override;
|
||||
|
||||
protected:
|
||||
SpeechSynthesisParent();
|
||||
virtual ~SpeechSynthesisParent();
|
||||
PSpeechSynthesisRequestParent* AllocPSpeechSynthesisRequestParent(const nsString& aText,
|
||||
const nsString& aLang,
|
||||
const nsString& aUri,
|
||||
const float& aVolume,
|
||||
const float& aRate,
|
||||
const float& aPitch)
|
||||
override;
|
||||
|
||||
bool DeallocPSpeechSynthesisRequestParent(PSpeechSynthesisRequestParent* aActor) override;
|
||||
|
||||
bool RecvPSpeechSynthesisRequestConstructor(PSpeechSynthesisRequestParent* aActor,
|
||||
const nsString& aText,
|
||||
const nsString& aLang,
|
||||
const nsString& aUri,
|
||||
const float& aVolume,
|
||||
const float& aRate,
|
||||
const float& aPitch) override;
|
||||
};
|
||||
|
||||
class SpeechSynthesisRequestParent : public PSpeechSynthesisRequestParent
|
||||
{
|
||||
public:
|
||||
explicit SpeechSynthesisRequestParent(SpeechTaskParent* aTask);
|
||||
virtual ~SpeechSynthesisRequestParent();
|
||||
|
||||
RefPtr<SpeechTaskParent> mTask;
|
||||
|
||||
protected:
|
||||
|
||||
void ActorDestroy(ActorDestroyReason aWhy) override;
|
||||
|
||||
bool RecvPause() override;
|
||||
|
||||
bool RecvResume() override;
|
||||
|
||||
bool RecvCancel() override;
|
||||
|
||||
bool RecvForceEnd() override;
|
||||
|
||||
bool RecvSetAudioOutputVolume(const float& aVolume) override;
|
||||
|
||||
bool Recv__delete__() override;
|
||||
};
|
||||
|
||||
class SpeechTaskParent : public nsSpeechTask
|
||||
{
|
||||
friend class SpeechSynthesisRequestParent;
|
||||
public:
|
||||
SpeechTaskParent(float aVolume, const nsAString& aUtterance)
|
||||
: nsSpeechTask(aVolume, aUtterance) {}
|
||||
|
||||
nsresult DispatchStartImpl(const nsAString& aUri);
|
||||
|
||||
nsresult DispatchEndImpl(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
nsresult DispatchPauseImpl(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
nsresult DispatchResumeImpl(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
nsresult DispatchErrorImpl(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
nsresult DispatchBoundaryImpl(const nsAString& aName,
|
||||
float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
nsresult DispatchMarkImpl(const nsAString& aName,
|
||||
float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
private:
|
||||
SpeechSynthesisRequestParent* mActor;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
70
dom/media/webspeech/synth/moz.build
Normal file
70
dom/media/webspeech/synth/moz.build
Normal file
|
|
@ -0,0 +1,70 @@
|
|||
# vim: set filetype=python:
|
||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
if CONFIG['MOZ_WEBSPEECH']:
|
||||
MOCHITEST_MANIFESTS += [
|
||||
'test/mochitest.ini',
|
||||
'test/startup/mochitest.ini',
|
||||
]
|
||||
|
||||
XPIDL_MODULE = 'dom_webspeechsynth'
|
||||
|
||||
XPIDL_SOURCES += [
|
||||
'nsISpeechService.idl',
|
||||
'nsISynthVoiceRegistry.idl'
|
||||
]
|
||||
|
||||
EXPORTS.mozilla.dom += [
|
||||
'ipc/SpeechSynthesisChild.h',
|
||||
'ipc/SpeechSynthesisParent.h',
|
||||
'nsSpeechTask.h',
|
||||
'nsSynthVoiceRegistry.h',
|
||||
'SpeechSynthesis.h',
|
||||
'SpeechSynthesisUtterance.h',
|
||||
'SpeechSynthesisVoice.h',
|
||||
]
|
||||
|
||||
UNIFIED_SOURCES += [
|
||||
'ipc/SpeechSynthesisChild.cpp',
|
||||
'ipc/SpeechSynthesisParent.cpp',
|
||||
'nsSpeechTask.cpp',
|
||||
'nsSynthVoiceRegistry.cpp',
|
||||
'SpeechSynthesis.cpp',
|
||||
'SpeechSynthesisUtterance.cpp',
|
||||
'SpeechSynthesisVoice.cpp',
|
||||
]
|
||||
|
||||
if CONFIG['MOZ_WEBSPEECH_TEST_BACKEND']:
|
||||
UNIFIED_SOURCES += [
|
||||
'test/FakeSynthModule.cpp',
|
||||
'test/nsFakeSynthServices.cpp'
|
||||
]
|
||||
|
||||
if CONFIG['MOZ_WIDGET_TOOLKIT'] == 'windows':
|
||||
DIRS += ['windows']
|
||||
|
||||
if CONFIG['MOZ_WIDGET_TOOLKIT'] == 'cocoa':
|
||||
DIRS += ['cocoa']
|
||||
|
||||
if CONFIG['MOZ_SYNTH_SPEECHD']:
|
||||
DIRS += ['speechd']
|
||||
|
||||
if CONFIG['MOZ_SYNTH_PICO']:
|
||||
DIRS += ['pico']
|
||||
|
||||
IPDL_SOURCES += [
|
||||
'ipc/PSpeechSynthesis.ipdl',
|
||||
'ipc/PSpeechSynthesisRequest.ipdl',
|
||||
]
|
||||
|
||||
include('/ipc/chromium/chromium-config.mozbuild')
|
||||
|
||||
FINAL_LIBRARY = 'xul'
|
||||
LOCAL_INCLUDES += [
|
||||
'ipc',
|
||||
]
|
||||
|
||||
if CONFIG['GNU_CXX']:
|
||||
CXXFLAGS += ['-Wno-error=shadow']
|
||||
173
dom/media/webspeech/synth/nsISpeechService.idl
Normal file
173
dom/media/webspeech/synth/nsISpeechService.idl
Normal file
|
|
@ -0,0 +1,173 @@
|
|||
/* -*- Mode: IDL; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
|
||||
* You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsISupports.idl"
|
||||
|
||||
typedef unsigned short SpeechServiceType;
|
||||
|
||||
/**
|
||||
* A callback is implemented by the service. For direct audio services, it is
|
||||
* required to implement these, although it could be helpful to use the
|
||||
* cancel method for shutting down the speech resources.
|
||||
*/
|
||||
[scriptable, uuid(c576de0c-8a3d-4570-be7e-9876d3e5bed2)]
|
||||
interface nsISpeechTaskCallback : nsISupports
|
||||
{
|
||||
/**
|
||||
* The user or application has paused the speech.
|
||||
*/
|
||||
void onPause();
|
||||
|
||||
/**
|
||||
* The user or application has resumed the speech.
|
||||
*/
|
||||
void onResume();
|
||||
|
||||
/**
|
||||
* The user or application has canceled the speech.
|
||||
*/
|
||||
void onCancel();
|
||||
|
||||
/**
|
||||
* The user or application has changed the volume of this speech.
|
||||
* This is only used on indirect audio service type.
|
||||
*/
|
||||
void onVolumeChanged(in float aVolume);
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* A task is associated with a single utterance. It is provided by the browser
|
||||
* to the service in the speak() method.
|
||||
*/
|
||||
[scriptable, builtinclass, uuid(ad59949c-2437-4b35-8eeb-d760caab75c5)]
|
||||
interface nsISpeechTask : nsISupports
|
||||
{
|
||||
/**
|
||||
* Prepare browser for speech.
|
||||
*
|
||||
* @param aCallback callback object for mid-speech operations.
|
||||
* @param aChannels number of audio channels. Only required
|
||||
* in direct audio services
|
||||
* @param aRate audio rate. Only required in direct audio services
|
||||
*/
|
||||
[optional_argc] void setup(in nsISpeechTaskCallback aCallback,
|
||||
[optional] in uint32_t aChannels,
|
||||
[optional] in uint32_t aRate);
|
||||
|
||||
/**
|
||||
* Send audio data to browser.
|
||||
*
|
||||
* @param aData an Int16Array with PCM-16 audio data.
|
||||
* @param aLandmarks an array of sample offset and landmark pairs.
|
||||
* Used for emiting boundary and mark events.
|
||||
*/
|
||||
[implicit_jscontext]
|
||||
void sendAudio(in jsval aData, in jsval aLandmarks);
|
||||
|
||||
[noscript]
|
||||
void sendAudioNative([array, size_is(aDataLen)] in short aData, in unsigned long aDataLen);
|
||||
|
||||
/**
|
||||
* Dispatch start event.
|
||||
*/
|
||||
void dispatchStart();
|
||||
|
||||
/**
|
||||
* Dispatch end event.
|
||||
*
|
||||
* @param aElapsedTime time in seconds since speech has started.
|
||||
* @param aCharIndex offset of spoken characters.
|
||||
*/
|
||||
void dispatchEnd(in float aElapsedTime, in unsigned long aCharIndex);
|
||||
|
||||
/**
|
||||
* Dispatch pause event.
|
||||
*
|
||||
* @param aElapsedTime time in seconds since speech has started.
|
||||
* @param aCharIndex offset of spoken characters.
|
||||
*/
|
||||
void dispatchPause(in float aElapsedTime, in unsigned long aCharIndex);
|
||||
|
||||
/**
|
||||
* Dispatch resume event.
|
||||
*
|
||||
* @param aElapsedTime time in seconds since speech has started.
|
||||
* @param aCharIndex offset of spoken characters.
|
||||
*/
|
||||
void dispatchResume(in float aElapsedTime, in unsigned long aCharIndex);
|
||||
|
||||
/**
|
||||
* Dispatch error event.
|
||||
*
|
||||
* @param aElapsedTime time in seconds since speech has started.
|
||||
* @param aCharIndex offset of spoken characters.
|
||||
*/
|
||||
void dispatchError(in float aElapsedTime, in unsigned long aCharIndex);
|
||||
|
||||
/**
|
||||
* Dispatch boundary event.
|
||||
*
|
||||
* @param aName name of boundary, 'word' or 'sentence'
|
||||
* @param aElapsedTime time in seconds since speech has started.
|
||||
* @param aCharIndex offset of spoken characters.
|
||||
*/
|
||||
void dispatchBoundary(in DOMString aName, in float aElapsedTime,
|
||||
in unsigned long aCharIndex);
|
||||
|
||||
/**
|
||||
* Dispatch mark event.
|
||||
*
|
||||
* @param aName mark identifier.
|
||||
* @param aElapsedTime time in seconds since speech has started.
|
||||
* @param aCharIndex offset of spoken characters.
|
||||
*/
|
||||
void dispatchMark(in DOMString aName, in float aElapsedTime, in unsigned long aCharIndex);
|
||||
};
|
||||
|
||||
/**
|
||||
* The main interface of a speech synthesis service.
|
||||
*
|
||||
* A service's speak method could be implemented in two ways:
|
||||
* 1. Indirect audio - the service is responsible for outputting audio.
|
||||
* The service calls the nsISpeechTask.dispatch* methods directly. Starting
|
||||
* with dispatchStart() and ending with dispatchEnd or dispatchError().
|
||||
*
|
||||
* 2. Direct audio - the service provides us with PCM-16 data, and we output it.
|
||||
* The service does not call the dispatch task methods directly. Instead,
|
||||
* audio information is provided at setup(), and audio data is sent with
|
||||
* sendAudio(). The utterance is terminated with an empty sendAudio().
|
||||
*/
|
||||
[scriptable, uuid(9b7d59db-88ff-43d0-b6ee-9f63d042d08f)]
|
||||
interface nsISpeechService : nsISupports
|
||||
{
|
||||
/**
|
||||
* Speak the given text using the voice identified byu the given uri. See
|
||||
* W3C Speech API spec for information about pitch and rate.
|
||||
* https://dvcs.w3.org/hg/speech-api/raw-file/tip/speechapi.html#utterance-attributes
|
||||
*
|
||||
* @param aText text to utter.
|
||||
* @param aUri unique voice identifier.
|
||||
* @param aVolume volume to speak voice in. Only relevant for indirect audio.
|
||||
* @param aRate rate to speak voice in.
|
||||
* @param aPitch pitch to speak voice in.
|
||||
* @param aTask task instance for utterance, used for sending events or audio
|
||||
* data back to browser.
|
||||
*/
|
||||
void speak(in DOMString aText, in DOMString aUri,
|
||||
in float aVolume, in float aRate, in float aPitch,
|
||||
in nsISpeechTask aTask);
|
||||
|
||||
const SpeechServiceType SERVICETYPE_DIRECT_AUDIO = 1;
|
||||
const SpeechServiceType SERVICETYPE_INDIRECT_AUDIO = 2;
|
||||
|
||||
readonly attribute SpeechServiceType serviceType;
|
||||
};
|
||||
|
||||
%{C++
|
||||
// This is the service category speech services could use to start up as
|
||||
// a component.
|
||||
#define NS_SPEECH_SYNTH_STARTED "speech-synth-started"
|
||||
%}
|
||||
77
dom/media/webspeech/synth/nsISynthVoiceRegistry.idl
Normal file
77
dom/media/webspeech/synth/nsISynthVoiceRegistry.idl
Normal file
|
|
@ -0,0 +1,77 @@
|
|||
/* -*- Mode: IDL; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
|
||||
* You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsISupports.idl"
|
||||
|
||||
interface nsISpeechService;
|
||||
|
||||
[scriptable, builtinclass, uuid(5d7a0b38-77e5-4ee5-897c-ce5db9b85d44)]
|
||||
interface nsISynthVoiceRegistry : nsISupports
|
||||
{
|
||||
/**
|
||||
* Register a speech synthesis voice.
|
||||
*
|
||||
* @param aService the service that provides this voice.
|
||||
* @param aUri a unique identifier for this voice.
|
||||
* @param aName human-readable name for this voice.
|
||||
* @param aLang a BCP 47 language tag.
|
||||
* @param aLocalService true if service does not require network.
|
||||
* @param aQueuesUtterances true if voice only speaks one utterance at a time
|
||||
*/
|
||||
void addVoice(in nsISpeechService aService, in DOMString aUri,
|
||||
in DOMString aName, in DOMString aLang,
|
||||
in boolean aLocalService, in boolean aQueuesUtterances);
|
||||
|
||||
/**
|
||||
* Remove a speech synthesis voice.
|
||||
*
|
||||
* @param aService the service that was used to add the voice.
|
||||
* @param aUri a unique identifier of an existing voice.
|
||||
*/
|
||||
void removeVoice(in nsISpeechService aService, in DOMString aUri);
|
||||
|
||||
/**
|
||||
* Notify content of voice availability changes. This allows content
|
||||
* to be notified of voice catalog changes in real time.
|
||||
*/
|
||||
void notifyVoicesChanged();
|
||||
|
||||
/**
|
||||
* Set a voice as default.
|
||||
*
|
||||
* @param aUri a unique identifier of an existing voice.
|
||||
* @param aIsDefault true if this voice should be toggled as default.
|
||||
*/
|
||||
void setDefaultVoice(in DOMString aUri, in boolean aIsDefault);
|
||||
|
||||
readonly attribute uint32_t voiceCount;
|
||||
|
||||
AString getVoice(in uint32_t aIndex);
|
||||
|
||||
bool isDefaultVoice(in DOMString aUri);
|
||||
|
||||
bool isLocalVoice(in DOMString aUri);
|
||||
|
||||
AString getVoiceLang(in DOMString aUri);
|
||||
|
||||
AString getVoiceName(in DOMString aUri);
|
||||
};
|
||||
|
||||
%{C++
|
||||
#define NS_SYNTHVOICEREGISTRY_CID \
|
||||
{ /* {7090524d-5574-4492-a77f-d8d558ced59d} */ \
|
||||
0x7090524d, \
|
||||
0x5574, \
|
||||
0x4492, \
|
||||
{ 0xa7, 0x7f, 0xd8, 0xd5, 0x58, 0xce, 0xd5, 0x9d } \
|
||||
}
|
||||
|
||||
#define NS_SYNTHVOICEREGISTRY_CONTRACTID \
|
||||
"@mozilla.org/synth-voice-registry;1"
|
||||
|
||||
#define NS_SYNTHVOICEREGISTRY_CLASSNAME \
|
||||
"Speech Synthesis Voice Registry"
|
||||
|
||||
%}
|
||||
783
dom/media/webspeech/synth/nsSpeechTask.cpp
Normal file
783
dom/media/webspeech/synth/nsSpeechTask.cpp
Normal file
|
|
@ -0,0 +1,783 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "AudioChannelAgent.h"
|
||||
#include "AudioChannelService.h"
|
||||
#include "AudioSegment.h"
|
||||
#include "MediaStreamListener.h"
|
||||
#include "nsSpeechTask.h"
|
||||
#include "nsSynthVoiceRegistry.h"
|
||||
#include "SharedBuffer.h"
|
||||
#include "SpeechSynthesis.h"
|
||||
|
||||
// GetCurrentTime is defined in winbase.h as zero argument macro forwarding to
|
||||
// GetTickCount() and conflicts with nsSpeechTask::GetCurrentTime().
|
||||
#ifdef GetCurrentTime
|
||||
#undef GetCurrentTime
|
||||
#endif
|
||||
|
||||
#undef LOG
|
||||
extern mozilla::LogModule* GetSpeechSynthLog();
|
||||
#define LOG(type, msg) MOZ_LOG(GetSpeechSynthLog(), type, msg)
|
||||
|
||||
#define AUDIO_TRACK 1
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class SynthStreamListener : public MediaStreamListener
|
||||
{
|
||||
public:
|
||||
explicit SynthStreamListener(nsSpeechTask* aSpeechTask,
|
||||
MediaStream* aStream) :
|
||||
mSpeechTask(aSpeechTask),
|
||||
mStream(aStream),
|
||||
mStarted(false)
|
||||
{
|
||||
}
|
||||
|
||||
void DoNotifyStarted()
|
||||
{
|
||||
if (mSpeechTask) {
|
||||
mSpeechTask->DispatchStartInner();
|
||||
}
|
||||
}
|
||||
|
||||
void DoNotifyFinished()
|
||||
{
|
||||
if (mSpeechTask) {
|
||||
mSpeechTask->DispatchEndInner(mSpeechTask->GetCurrentTime(),
|
||||
mSpeechTask->GetCurrentCharOffset());
|
||||
}
|
||||
}
|
||||
|
||||
void NotifyEvent(MediaStreamGraph* aGraph,
|
||||
MediaStreamGraphEvent event) override
|
||||
{
|
||||
switch (event) {
|
||||
case MediaStreamGraphEvent::EVENT_FINISHED:
|
||||
{
|
||||
if (!mStarted) {
|
||||
mStarted = true;
|
||||
nsCOMPtr<nsIRunnable> startRunnable =
|
||||
NewRunnableMethod(this, &SynthStreamListener::DoNotifyStarted);
|
||||
aGraph->DispatchToMainThreadAfterStreamStateUpdate(startRunnable.forget());
|
||||
}
|
||||
|
||||
nsCOMPtr<nsIRunnable> endRunnable =
|
||||
NewRunnableMethod(this, &SynthStreamListener::DoNotifyFinished);
|
||||
aGraph->DispatchToMainThreadAfterStreamStateUpdate(endRunnable.forget());
|
||||
}
|
||||
break;
|
||||
case MediaStreamGraphEvent::EVENT_REMOVED:
|
||||
mSpeechTask = nullptr;
|
||||
// Dereference MediaStream to destroy safety
|
||||
mStream = nullptr;
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
void NotifyBlockingChanged(MediaStreamGraph* aGraph, Blocking aBlocked) override
|
||||
{
|
||||
if (aBlocked == MediaStreamListener::UNBLOCKED && !mStarted) {
|
||||
mStarted = true;
|
||||
nsCOMPtr<nsIRunnable> event =
|
||||
NewRunnableMethod(this, &SynthStreamListener::DoNotifyStarted);
|
||||
aGraph->DispatchToMainThreadAfterStreamStateUpdate(event.forget());
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Raw pointer; if we exist, the stream exists,
|
||||
// and 'mSpeechTask' exclusively owns it and therefor exists as well.
|
||||
nsSpeechTask* mSpeechTask;
|
||||
// This is KungFuDeathGrip for MediaStream
|
||||
RefPtr<MediaStream> mStream;
|
||||
|
||||
bool mStarted;
|
||||
};
|
||||
|
||||
// nsSpeechTask
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION(nsSpeechTask, mSpeechSynthesis, mUtterance, mCallback);
|
||||
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(nsSpeechTask)
|
||||
NS_INTERFACE_MAP_ENTRY(nsISpeechTask)
|
||||
NS_INTERFACE_MAP_ENTRY(nsIAudioChannelAgentCallback)
|
||||
NS_INTERFACE_MAP_ENTRY(nsISupportsWeakReference)
|
||||
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsISpeechTask)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTING_ADDREF(nsSpeechTask)
|
||||
NS_IMPL_CYCLE_COLLECTING_RELEASE(nsSpeechTask)
|
||||
|
||||
nsSpeechTask::nsSpeechTask(SpeechSynthesisUtterance* aUtterance)
|
||||
: mUtterance(aUtterance)
|
||||
, mInited(false)
|
||||
, mPrePaused(false)
|
||||
, mPreCanceled(false)
|
||||
, mCallback(nullptr)
|
||||
, mIndirectAudio(false)
|
||||
{
|
||||
mText = aUtterance->mText;
|
||||
mVolume = aUtterance->Volume();
|
||||
}
|
||||
|
||||
nsSpeechTask::nsSpeechTask(float aVolume, const nsAString& aText)
|
||||
: mUtterance(nullptr)
|
||||
, mVolume(aVolume)
|
||||
, mText(aText)
|
||||
, mInited(false)
|
||||
, mPrePaused(false)
|
||||
, mPreCanceled(false)
|
||||
, mCallback(nullptr)
|
||||
, mIndirectAudio(false)
|
||||
{
|
||||
}
|
||||
|
||||
nsSpeechTask::~nsSpeechTask()
|
||||
{
|
||||
LOG(LogLevel::Debug, ("~nsSpeechTask"));
|
||||
if (mStream) {
|
||||
if (!mStream->IsDestroyed()) {
|
||||
mStream->Destroy();
|
||||
}
|
||||
|
||||
// This will finally destroyed by SynthStreamListener becasue
|
||||
// MediaStream::Destroy() is async.
|
||||
mStream = nullptr;
|
||||
}
|
||||
|
||||
if (mPort) {
|
||||
mPort->Destroy();
|
||||
mPort = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::InitDirectAudio()
|
||||
{
|
||||
mStream = MediaStreamGraph::GetInstance(MediaStreamGraph::AUDIO_THREAD_DRIVER,
|
||||
AudioChannel::Normal)->
|
||||
CreateSourceStream();
|
||||
mIndirectAudio = false;
|
||||
mInited = true;
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::InitIndirectAudio()
|
||||
{
|
||||
mIndirectAudio = true;
|
||||
mInited = true;
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::SetChosenVoiceURI(const nsAString& aUri)
|
||||
{
|
||||
mChosenVoiceURI = aUri;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::Setup(nsISpeechTaskCallback* aCallback,
|
||||
uint32_t aChannels, uint32_t aRate, uint8_t argc)
|
||||
{
|
||||
MOZ_ASSERT(XRE_IsParentProcess());
|
||||
|
||||
LOG(LogLevel::Debug, ("nsSpeechTask::Setup"));
|
||||
|
||||
mCallback = aCallback;
|
||||
|
||||
if (mIndirectAudio) {
|
||||
MOZ_ASSERT(!mStream);
|
||||
if (argc > 0) {
|
||||
NS_WARNING("Audio info arguments in Setup() are ignored for indirect audio services.");
|
||||
}
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
// mStream is set up in Init() that should be called before this.
|
||||
MOZ_ASSERT(mStream);
|
||||
|
||||
mStream->AddListener(new SynthStreamListener(this, mStream));
|
||||
|
||||
// XXX: Support more than one channel
|
||||
if(NS_WARN_IF(!(aChannels == 1))) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
mChannels = aChannels;
|
||||
|
||||
AudioSegment* segment = new AudioSegment();
|
||||
mStream->AddAudioTrack(AUDIO_TRACK, aRate, 0, segment);
|
||||
mStream->AddAudioOutput(this);
|
||||
mStream->SetAudioOutputVolume(this, mVolume);
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
static RefPtr<mozilla::SharedBuffer>
|
||||
makeSamples(int16_t* aData, uint32_t aDataLen)
|
||||
{
|
||||
RefPtr<mozilla::SharedBuffer> samples =
|
||||
SharedBuffer::Create(aDataLen * sizeof(int16_t));
|
||||
int16_t* frames = static_cast<int16_t*>(samples->Data());
|
||||
|
||||
for (uint32_t i = 0; i < aDataLen; i++) {
|
||||
frames[i] = aData[i];
|
||||
}
|
||||
|
||||
return samples;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::SendAudio(JS::Handle<JS::Value> aData, JS::Handle<JS::Value> aLandmarks,
|
||||
JSContext* aCx)
|
||||
{
|
||||
MOZ_ASSERT(XRE_IsParentProcess());
|
||||
|
||||
if(NS_WARN_IF(!(mStream))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
if(NS_WARN_IF(mStream->IsDestroyed())) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
if(NS_WARN_IF(!(mChannels))) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
if(NS_WARN_IF(!(aData.isObject()))) {
|
||||
return NS_ERROR_INVALID_ARG;
|
||||
}
|
||||
|
||||
if (mIndirectAudio) {
|
||||
NS_WARNING("Can't call SendAudio from an indirect audio speech service.");
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
JS::Rooted<JSObject*> darray(aCx, &aData.toObject());
|
||||
JSAutoCompartment ac(aCx, darray);
|
||||
|
||||
JS::Rooted<JSObject*> tsrc(aCx, nullptr);
|
||||
|
||||
// Allow either Int16Array or plain JS Array
|
||||
if (JS_IsInt16Array(darray)) {
|
||||
tsrc = darray;
|
||||
} else {
|
||||
bool isArray;
|
||||
if (!JS_IsArrayObject(aCx, darray, &isArray)) {
|
||||
return NS_ERROR_UNEXPECTED;
|
||||
}
|
||||
if (isArray) {
|
||||
tsrc = JS_NewInt16ArrayFromArray(aCx, darray);
|
||||
}
|
||||
}
|
||||
|
||||
if (!tsrc) {
|
||||
return NS_ERROR_DOM_TYPE_MISMATCH_ERR;
|
||||
}
|
||||
|
||||
uint32_t dataLen = JS_GetTypedArrayLength(tsrc);
|
||||
RefPtr<mozilla::SharedBuffer> samples;
|
||||
{
|
||||
JS::AutoCheckCannotGC nogc;
|
||||
bool isShared;
|
||||
int16_t* data = JS_GetInt16ArrayData(tsrc, &isShared, nogc);
|
||||
if (isShared) {
|
||||
// Must opt in to using shared data.
|
||||
return NS_ERROR_DOM_TYPE_MISMATCH_ERR;
|
||||
}
|
||||
samples = makeSamples(data, dataLen);
|
||||
}
|
||||
SendAudioImpl(samples, dataLen);
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::SendAudioNative(int16_t* aData, uint32_t aDataLen)
|
||||
{
|
||||
MOZ_ASSERT(XRE_IsParentProcess());
|
||||
|
||||
if(NS_WARN_IF(!(mStream))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
if(NS_WARN_IF(mStream->IsDestroyed())) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
if(NS_WARN_IF(!(mChannels))) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
if (mIndirectAudio) {
|
||||
NS_WARNING("Can't call SendAudio from an indirect audio speech service.");
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
RefPtr<mozilla::SharedBuffer> samples = makeSamples(aData, aDataLen);
|
||||
SendAudioImpl(samples, aDataLen);
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::SendAudioImpl(RefPtr<mozilla::SharedBuffer>& aSamples, uint32_t aDataLen)
|
||||
{
|
||||
if (aDataLen == 0) {
|
||||
mStream->EndAllTrackAndFinish();
|
||||
return;
|
||||
}
|
||||
|
||||
AudioSegment segment;
|
||||
AutoTArray<const int16_t*, 1> channelData;
|
||||
channelData.AppendElement(static_cast<int16_t*>(aSamples->Data()));
|
||||
segment.AppendFrames(aSamples.forget(), channelData, aDataLen,
|
||||
PRINCIPAL_HANDLE_NONE);
|
||||
mStream->AppendToTrack(1, &segment);
|
||||
mStream->AdvanceKnownTracksTime(STREAM_TIME_MAX);
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::DispatchStart()
|
||||
{
|
||||
if (!mIndirectAudio) {
|
||||
NS_WARNING("Can't call DispatchStart() from a direct audio speech service");
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return DispatchStartInner();
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSpeechTask::DispatchStartInner()
|
||||
{
|
||||
nsSynthVoiceRegistry::GetInstance()->SetIsSpeaking(true);
|
||||
return DispatchStartImpl();
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSpeechTask::DispatchStartImpl()
|
||||
{
|
||||
return DispatchStartImpl(mChosenVoiceURI);
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSpeechTask::DispatchStartImpl(const nsAString& aUri)
|
||||
{
|
||||
LOG(LogLevel::Debug, ("nsSpeechTask::DispatchStart"));
|
||||
|
||||
MOZ_ASSERT(mUtterance);
|
||||
if(NS_WARN_IF(!(mUtterance->mState == SpeechSynthesisUtterance::STATE_PENDING))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
CreateAudioChannelAgent();
|
||||
|
||||
mUtterance->mState = SpeechSynthesisUtterance::STATE_SPEAKING;
|
||||
mUtterance->mChosenVoiceURI = aUri;
|
||||
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("start"), 0, 0,
|
||||
EmptyString());
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::DispatchEnd(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
if (!mIndirectAudio) {
|
||||
NS_WARNING("Can't call DispatchEnd() from a direct audio speech service");
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return DispatchEndInner(aElapsedTime, aCharIndex);
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSpeechTask::DispatchEndInner(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
if (!mPreCanceled) {
|
||||
nsSynthVoiceRegistry::GetInstance()->SpeakNext();
|
||||
}
|
||||
|
||||
return DispatchEndImpl(aElapsedTime, aCharIndex);
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSpeechTask::DispatchEndImpl(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
LOG(LogLevel::Debug, ("nsSpeechTask::DispatchEnd\n"));
|
||||
|
||||
DestroyAudioChannelAgent();
|
||||
|
||||
MOZ_ASSERT(mUtterance);
|
||||
if(NS_WARN_IF(mUtterance->mState == SpeechSynthesisUtterance::STATE_ENDED)) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
// XXX: This should not be here, but it prevents a crash in MSG.
|
||||
if (mStream) {
|
||||
mStream->Destroy();
|
||||
}
|
||||
|
||||
RefPtr<SpeechSynthesisUtterance> utterance = mUtterance;
|
||||
|
||||
if (mSpeechSynthesis) {
|
||||
mSpeechSynthesis->OnEnd(this);
|
||||
}
|
||||
|
||||
if (utterance->mState == SpeechSynthesisUtterance::STATE_PENDING) {
|
||||
utterance->mState = SpeechSynthesisUtterance::STATE_NONE;
|
||||
} else {
|
||||
utterance->mState = SpeechSynthesisUtterance::STATE_ENDED;
|
||||
utterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("end"),
|
||||
aCharIndex, aElapsedTime,
|
||||
EmptyString());
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::DispatchPause(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
if (!mIndirectAudio) {
|
||||
NS_WARNING("Can't call DispatchPause() from a direct audio speech service");
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return DispatchPauseImpl(aElapsedTime, aCharIndex);
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSpeechTask::DispatchPauseImpl(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
LOG(LogLevel::Debug, ("nsSpeechTask::DispatchPause"));
|
||||
MOZ_ASSERT(mUtterance);
|
||||
if(NS_WARN_IF(mUtterance->mPaused)) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
if(NS_WARN_IF(mUtterance->mState == SpeechSynthesisUtterance::STATE_ENDED)) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
mUtterance->mPaused = true;
|
||||
if (mUtterance->mState == SpeechSynthesisUtterance::STATE_SPEAKING) {
|
||||
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("pause"),
|
||||
aCharIndex, aElapsedTime,
|
||||
EmptyString());
|
||||
}
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::DispatchResume(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
if (!mIndirectAudio) {
|
||||
NS_WARNING("Can't call DispatchResume() from a direct audio speech service");
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return DispatchResumeImpl(aElapsedTime, aCharIndex);
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSpeechTask::DispatchResumeImpl(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
LOG(LogLevel::Debug, ("nsSpeechTask::DispatchResume"));
|
||||
MOZ_ASSERT(mUtterance);
|
||||
if(NS_WARN_IF(!(mUtterance->mPaused))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
if(NS_WARN_IF(mUtterance->mState == SpeechSynthesisUtterance::STATE_ENDED)) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
mUtterance->mPaused = false;
|
||||
if (mUtterance->mState == SpeechSynthesisUtterance::STATE_SPEAKING) {
|
||||
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("resume"),
|
||||
aCharIndex, aElapsedTime,
|
||||
EmptyString());
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::DispatchError(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
LOG(LogLevel::Debug, ("nsSpeechTask::DispatchError"));
|
||||
|
||||
if (!mIndirectAudio) {
|
||||
NS_WARNING("Can't call DispatchError() from a direct audio speech service");
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
if (!mPreCanceled) {
|
||||
nsSynthVoiceRegistry::GetInstance()->SpeakNext();
|
||||
}
|
||||
|
||||
return DispatchErrorImpl(aElapsedTime, aCharIndex);
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSpeechTask::DispatchErrorImpl(float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
MOZ_ASSERT(mUtterance);
|
||||
if(NS_WARN_IF(mUtterance->mState == SpeechSynthesisUtterance::STATE_ENDED)) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
if (mSpeechSynthesis) {
|
||||
mSpeechSynthesis->OnEnd(this);
|
||||
}
|
||||
|
||||
mUtterance->mState = SpeechSynthesisUtterance::STATE_ENDED;
|
||||
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("error"),
|
||||
aCharIndex, aElapsedTime,
|
||||
EmptyString());
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::DispatchBoundary(const nsAString& aName,
|
||||
float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
if (!mIndirectAudio) {
|
||||
NS_WARNING("Can't call DispatchBoundary() from a direct audio speech service");
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return DispatchBoundaryImpl(aName, aElapsedTime, aCharIndex);
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSpeechTask::DispatchBoundaryImpl(const nsAString& aName,
|
||||
float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
MOZ_ASSERT(mUtterance);
|
||||
if(NS_WARN_IF(!(mUtterance->mState == SpeechSynthesisUtterance::STATE_SPEAKING))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("boundary"),
|
||||
aCharIndex, aElapsedTime,
|
||||
aName);
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::DispatchMark(const nsAString& aName,
|
||||
float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
if (!mIndirectAudio) {
|
||||
NS_WARNING("Can't call DispatchMark() from a direct audio speech service");
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return DispatchMarkImpl(aName, aElapsedTime, aCharIndex);
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSpeechTask::DispatchMarkImpl(const nsAString& aName,
|
||||
float aElapsedTime, uint32_t aCharIndex)
|
||||
{
|
||||
MOZ_ASSERT(mUtterance);
|
||||
if(NS_WARN_IF(!(mUtterance->mState == SpeechSynthesisUtterance::STATE_SPEAKING))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("mark"),
|
||||
aCharIndex, aElapsedTime,
|
||||
aName);
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::Pause()
|
||||
{
|
||||
MOZ_ASSERT(XRE_IsParentProcess());
|
||||
|
||||
if (mCallback) {
|
||||
DebugOnly<nsresult> rv = mCallback->OnPause();
|
||||
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv), "Unable to call onPause() callback");
|
||||
}
|
||||
|
||||
if (mStream) {
|
||||
mStream->Suspend();
|
||||
}
|
||||
|
||||
if (!mInited) {
|
||||
mPrePaused = true;
|
||||
}
|
||||
|
||||
if (!mIndirectAudio) {
|
||||
DispatchPauseImpl(GetCurrentTime(), GetCurrentCharOffset());
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::Resume()
|
||||
{
|
||||
MOZ_ASSERT(XRE_IsParentProcess());
|
||||
|
||||
if (mCallback) {
|
||||
DebugOnly<nsresult> rv = mCallback->OnResume();
|
||||
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv),
|
||||
"Unable to call onResume() callback");
|
||||
}
|
||||
|
||||
if (mStream) {
|
||||
mStream->Resume();
|
||||
}
|
||||
|
||||
if (mPrePaused) {
|
||||
mPrePaused = false;
|
||||
nsSynthVoiceRegistry::GetInstance()->ResumeQueue();
|
||||
}
|
||||
|
||||
if (!mIndirectAudio) {
|
||||
DispatchResumeImpl(GetCurrentTime(), GetCurrentCharOffset());
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::Cancel()
|
||||
{
|
||||
MOZ_ASSERT(XRE_IsParentProcess());
|
||||
|
||||
LOG(LogLevel::Debug, ("nsSpeechTask::Cancel"));
|
||||
|
||||
if (mCallback) {
|
||||
DebugOnly<nsresult> rv = mCallback->OnCancel();
|
||||
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv),
|
||||
"Unable to call onCancel() callback");
|
||||
}
|
||||
|
||||
if (mStream) {
|
||||
mStream->Suspend();
|
||||
}
|
||||
|
||||
if (!mInited) {
|
||||
mPreCanceled = true;
|
||||
}
|
||||
|
||||
if (!mIndirectAudio) {
|
||||
DispatchEndInner(GetCurrentTime(), GetCurrentCharOffset());
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::ForceEnd()
|
||||
{
|
||||
if (mStream) {
|
||||
mStream->Suspend();
|
||||
}
|
||||
|
||||
if (!mInited) {
|
||||
mPreCanceled = true;
|
||||
}
|
||||
|
||||
DispatchEndInner(GetCurrentTime(), GetCurrentCharOffset());
|
||||
}
|
||||
|
||||
float
|
||||
nsSpeechTask::GetCurrentTime()
|
||||
{
|
||||
return mStream ? (float)(mStream->GetCurrentTime() / 1000000.0) : 0;
|
||||
}
|
||||
|
||||
uint32_t
|
||||
nsSpeechTask::GetCurrentCharOffset()
|
||||
{
|
||||
return mStream && mStream->IsFinished() ? mText.Length() : 0;
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::SetSpeechSynthesis(SpeechSynthesis* aSpeechSynthesis)
|
||||
{
|
||||
mSpeechSynthesis = aSpeechSynthesis;
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::CreateAudioChannelAgent()
|
||||
{
|
||||
if (!mUtterance) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (mAudioChannelAgent) {
|
||||
mAudioChannelAgent->NotifyStoppedPlaying();
|
||||
}
|
||||
|
||||
mAudioChannelAgent = new AudioChannelAgent();
|
||||
mAudioChannelAgent->InitWithWeakCallback(mUtterance->GetOwner(),
|
||||
static_cast<int32_t>(AudioChannelService::GetDefaultAudioChannel()),
|
||||
this);
|
||||
|
||||
AudioPlaybackConfig config;
|
||||
nsresult rv = mAudioChannelAgent->NotifyStartedPlaying(&config,
|
||||
AudioChannelService::AudibleState::eAudible);
|
||||
if (NS_WARN_IF(NS_FAILED(rv))) {
|
||||
return;
|
||||
}
|
||||
|
||||
WindowVolumeChanged(config.mVolume, config.mMuted);
|
||||
WindowSuspendChanged(config.mSuspend);
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::DestroyAudioChannelAgent()
|
||||
{
|
||||
if (mAudioChannelAgent) {
|
||||
mAudioChannelAgent->NotifyStoppedPlaying();
|
||||
mAudioChannelAgent = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::WindowVolumeChanged(float aVolume, bool aMuted)
|
||||
{
|
||||
SetAudioOutputVolume(aMuted ? 0.0 : mVolume * aVolume);
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::WindowSuspendChanged(nsSuspendedTypes aSuspend)
|
||||
{
|
||||
if (!mUtterance) {
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
if (aSuspend == nsISuspendedTypes::NONE_SUSPENDED &&
|
||||
mUtterance->mPaused) {
|
||||
Resume();
|
||||
} else if (aSuspend != nsISuspendedTypes::NONE_SUSPENDED &&
|
||||
!mUtterance->mPaused) {
|
||||
Pause();
|
||||
}
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSpeechTask::WindowAudioCaptureChanged(bool aCapture)
|
||||
{
|
||||
// This is not supported yet.
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
void
|
||||
nsSpeechTask::SetAudioOutputVolume(float aVolume)
|
||||
{
|
||||
if (mStream && !mStream->IsDestroyed()) {
|
||||
mStream->SetAudioOutputVolume(this, aVolume);
|
||||
}
|
||||
if (mIndirectAudio) {
|
||||
mCallback->OnVolumeChanged(aVolume);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
139
dom/media/webspeech/synth/nsSpeechTask.h
Normal file
139
dom/media/webspeech/synth/nsSpeechTask.h
Normal file
|
|
@ -0,0 +1,139 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_nsSpeechTask_h
|
||||
#define mozilla_dom_nsSpeechTask_h
|
||||
|
||||
#include "MediaStreamGraph.h"
|
||||
#include "SpeechSynthesisUtterance.h"
|
||||
#include "nsIAudioChannelAgent.h"
|
||||
#include "nsISpeechService.h"
|
||||
|
||||
namespace mozilla {
|
||||
|
||||
class SharedBuffer;
|
||||
|
||||
namespace dom {
|
||||
|
||||
class SpeechSynthesisUtterance;
|
||||
class SpeechSynthesis;
|
||||
class SynthStreamListener;
|
||||
|
||||
class nsSpeechTask : public nsISpeechTask
|
||||
, public nsIAudioChannelAgentCallback
|
||||
, public nsSupportsWeakReference
|
||||
{
|
||||
friend class SynthStreamListener;
|
||||
|
||||
public:
|
||||
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
|
||||
NS_DECL_CYCLE_COLLECTION_CLASS_AMBIGUOUS(nsSpeechTask, nsISpeechTask)
|
||||
|
||||
NS_DECL_NSISPEECHTASK
|
||||
NS_DECL_NSIAUDIOCHANNELAGENTCALLBACK
|
||||
|
||||
explicit nsSpeechTask(SpeechSynthesisUtterance* aUtterance);
|
||||
nsSpeechTask(float aVolume, const nsAString& aText);
|
||||
|
||||
virtual void Pause();
|
||||
|
||||
virtual void Resume();
|
||||
|
||||
virtual void Cancel();
|
||||
|
||||
virtual void ForceEnd();
|
||||
|
||||
float GetCurrentTime();
|
||||
|
||||
uint32_t GetCurrentCharOffset();
|
||||
|
||||
void SetSpeechSynthesis(SpeechSynthesis* aSpeechSynthesis);
|
||||
|
||||
void InitDirectAudio();
|
||||
void InitIndirectAudio();
|
||||
|
||||
void SetChosenVoiceURI(const nsAString& aUri);
|
||||
|
||||
virtual void SetAudioOutputVolume(float aVolume);
|
||||
|
||||
bool IsPreCanceled()
|
||||
{
|
||||
return mPreCanceled;
|
||||
};
|
||||
|
||||
bool IsPrePaused()
|
||||
{
|
||||
return mPrePaused;
|
||||
}
|
||||
|
||||
protected:
|
||||
virtual ~nsSpeechTask();
|
||||
|
||||
nsresult DispatchStartImpl();
|
||||
|
||||
virtual nsresult DispatchStartImpl(const nsAString& aUri);
|
||||
|
||||
virtual nsresult DispatchEndImpl(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
virtual nsresult DispatchPauseImpl(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
virtual nsresult DispatchResumeImpl(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
virtual nsresult DispatchErrorImpl(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
virtual nsresult DispatchBoundaryImpl(const nsAString& aName,
|
||||
float aElapsedTime,
|
||||
uint32_t aCharIndex);
|
||||
|
||||
virtual nsresult DispatchMarkImpl(const nsAString& aName,
|
||||
float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
RefPtr<SpeechSynthesisUtterance> mUtterance;
|
||||
|
||||
float mVolume;
|
||||
|
||||
nsString mText;
|
||||
|
||||
bool mInited;
|
||||
|
||||
bool mPrePaused;
|
||||
|
||||
bool mPreCanceled;
|
||||
|
||||
private:
|
||||
void End();
|
||||
|
||||
void SendAudioImpl(RefPtr<mozilla::SharedBuffer>& aSamples, uint32_t aDataLen);
|
||||
|
||||
nsresult DispatchStartInner();
|
||||
|
||||
nsresult DispatchEndInner(float aElapsedTime, uint32_t aCharIndex);
|
||||
|
||||
void CreateAudioChannelAgent();
|
||||
|
||||
void DestroyAudioChannelAgent();
|
||||
|
||||
RefPtr<SourceMediaStream> mStream;
|
||||
|
||||
RefPtr<MediaInputPort> mPort;
|
||||
|
||||
nsCOMPtr<nsISpeechTaskCallback> mCallback;
|
||||
|
||||
nsCOMPtr<nsIAudioChannelAgent> mAudioChannelAgent;
|
||||
|
||||
uint32_t mChannels;
|
||||
|
||||
RefPtr<SpeechSynthesis> mSpeechSynthesis;
|
||||
|
||||
bool mIndirectAudio;
|
||||
|
||||
nsString mChosenVoiceURI;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
835
dom/media/webspeech/synth/nsSynthVoiceRegistry.cpp
Normal file
835
dom/media/webspeech/synth/nsSynthVoiceRegistry.cpp
Normal file
|
|
@ -0,0 +1,835 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsILocaleService.h"
|
||||
#include "nsISpeechService.h"
|
||||
#include "nsServiceManagerUtils.h"
|
||||
#include "nsCategoryManagerUtils.h"
|
||||
|
||||
#include "MediaPrefs.h"
|
||||
#include "SpeechSynthesisUtterance.h"
|
||||
#include "SpeechSynthesisVoice.h"
|
||||
#include "nsSynthVoiceRegistry.h"
|
||||
#include "nsSpeechTask.h"
|
||||
#include "AudioChannelService.h"
|
||||
|
||||
#include "nsString.h"
|
||||
#include "mozilla/StaticPtr.h"
|
||||
#include "mozilla/dom/ContentChild.h"
|
||||
#include "mozilla/dom/ContentParent.h"
|
||||
#include "mozilla/Unused.h"
|
||||
|
||||
#include "SpeechSynthesisChild.h"
|
||||
#include "SpeechSynthesisParent.h"
|
||||
|
||||
#undef LOG
|
||||
extern mozilla::LogModule* GetSpeechSynthLog();
|
||||
#define LOG(type, msg) MOZ_LOG(GetSpeechSynthLog(), type, msg)
|
||||
|
||||
namespace {
|
||||
|
||||
void
|
||||
GetAllSpeechSynthActors(InfallibleTArray<mozilla::dom::SpeechSynthesisParent*>& aActors)
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
MOZ_ASSERT(aActors.IsEmpty());
|
||||
|
||||
AutoTArray<mozilla::dom::ContentParent*, 20> contentActors;
|
||||
mozilla::dom::ContentParent::GetAll(contentActors);
|
||||
|
||||
for (uint32_t contentIndex = 0;
|
||||
contentIndex < contentActors.Length();
|
||||
++contentIndex) {
|
||||
MOZ_ASSERT(contentActors[contentIndex]);
|
||||
|
||||
AutoTArray<mozilla::dom::PSpeechSynthesisParent*, 5> speechsynthActors;
|
||||
contentActors[contentIndex]->ManagedPSpeechSynthesisParent(speechsynthActors);
|
||||
|
||||
for (uint32_t speechsynthIndex = 0;
|
||||
speechsynthIndex < speechsynthActors.Length();
|
||||
++speechsynthIndex) {
|
||||
MOZ_ASSERT(speechsynthActors[speechsynthIndex]);
|
||||
|
||||
mozilla::dom::SpeechSynthesisParent* actor =
|
||||
static_cast<mozilla::dom::SpeechSynthesisParent*>(speechsynthActors[speechsynthIndex]);
|
||||
aActors.AppendElement(actor);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
// VoiceData
|
||||
|
||||
class VoiceData final
|
||||
{
|
||||
private:
|
||||
// Private destructor, to discourage deletion outside of Release():
|
||||
~VoiceData() {}
|
||||
|
||||
public:
|
||||
VoiceData(nsISpeechService* aService, const nsAString& aUri,
|
||||
const nsAString& aName, const nsAString& aLang,
|
||||
bool aIsLocal, bool aQueuesUtterances)
|
||||
: mService(aService)
|
||||
, mUri(aUri)
|
||||
, mName(aName)
|
||||
, mLang(aLang)
|
||||
, mIsLocal(aIsLocal)
|
||||
, mIsQueued(aQueuesUtterances) {}
|
||||
|
||||
NS_INLINE_DECL_REFCOUNTING(VoiceData)
|
||||
|
||||
nsCOMPtr<nsISpeechService> mService;
|
||||
|
||||
nsString mUri;
|
||||
|
||||
nsString mName;
|
||||
|
||||
nsString mLang;
|
||||
|
||||
bool mIsLocal;
|
||||
|
||||
bool mIsQueued;
|
||||
};
|
||||
|
||||
// GlobalQueueItem
|
||||
|
||||
class GlobalQueueItem final
|
||||
{
|
||||
private:
|
||||
// Private destructor, to discourage deletion outside of Release():
|
||||
~GlobalQueueItem() {}
|
||||
|
||||
public:
|
||||
GlobalQueueItem(VoiceData* aVoice, nsSpeechTask* aTask, const nsAString& aText,
|
||||
const float& aVolume, const float& aRate, const float& aPitch)
|
||||
: mVoice(aVoice)
|
||||
, mTask(aTask)
|
||||
, mText(aText)
|
||||
, mVolume(aVolume)
|
||||
, mRate(aRate)
|
||||
, mPitch(aPitch) {}
|
||||
|
||||
NS_INLINE_DECL_REFCOUNTING(GlobalQueueItem)
|
||||
|
||||
RefPtr<VoiceData> mVoice;
|
||||
|
||||
RefPtr<nsSpeechTask> mTask;
|
||||
|
||||
nsString mText;
|
||||
|
||||
float mVolume;
|
||||
|
||||
float mRate;
|
||||
|
||||
float mPitch;
|
||||
|
||||
bool mIsLocal;
|
||||
};
|
||||
|
||||
// nsSynthVoiceRegistry
|
||||
|
||||
static StaticRefPtr<nsSynthVoiceRegistry> gSynthVoiceRegistry;
|
||||
|
||||
NS_IMPL_ISUPPORTS(nsSynthVoiceRegistry, nsISynthVoiceRegistry)
|
||||
|
||||
nsSynthVoiceRegistry::nsSynthVoiceRegistry()
|
||||
: mSpeechSynthChild(nullptr)
|
||||
, mUseGlobalQueue(false)
|
||||
, mIsSpeaking(false)
|
||||
{
|
||||
if (XRE_IsContentProcess()) {
|
||||
|
||||
mSpeechSynthChild = new SpeechSynthesisChild();
|
||||
ContentChild::GetSingleton()->SendPSpeechSynthesisConstructor(mSpeechSynthChild);
|
||||
|
||||
InfallibleTArray<RemoteVoice> voices;
|
||||
InfallibleTArray<nsString> defaults;
|
||||
bool isSpeaking;
|
||||
|
||||
mSpeechSynthChild->SendReadVoicesAndState(&voices, &defaults, &isSpeaking);
|
||||
|
||||
for (uint32_t i = 0; i < voices.Length(); ++i) {
|
||||
RemoteVoice voice = voices[i];
|
||||
AddVoiceImpl(nullptr, voice.voiceURI(),
|
||||
voice.name(), voice.lang(),
|
||||
voice.localService(), voice.queued());
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < defaults.Length(); ++i) {
|
||||
SetDefaultVoice(defaults[i], true);
|
||||
}
|
||||
|
||||
mIsSpeaking = isSpeaking;
|
||||
}
|
||||
}
|
||||
|
||||
nsSynthVoiceRegistry::~nsSynthVoiceRegistry()
|
||||
{
|
||||
LOG(LogLevel::Debug, ("~nsSynthVoiceRegistry"));
|
||||
|
||||
// mSpeechSynthChild's lifecycle is managed by the Content protocol.
|
||||
mSpeechSynthChild = nullptr;
|
||||
|
||||
mUriVoiceMap.Clear();
|
||||
}
|
||||
|
||||
nsSynthVoiceRegistry*
|
||||
nsSynthVoiceRegistry::GetInstance()
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
|
||||
if (!gSynthVoiceRegistry) {
|
||||
gSynthVoiceRegistry = new nsSynthVoiceRegistry();
|
||||
if (XRE_IsParentProcess()) {
|
||||
// Start up all speech synth services.
|
||||
NS_CreateServicesFromCategory(NS_SPEECH_SYNTH_STARTED, nullptr,
|
||||
NS_SPEECH_SYNTH_STARTED);
|
||||
}
|
||||
}
|
||||
|
||||
return gSynthVoiceRegistry;
|
||||
}
|
||||
|
||||
already_AddRefed<nsSynthVoiceRegistry>
|
||||
nsSynthVoiceRegistry::GetInstanceForService()
|
||||
{
|
||||
RefPtr<nsSynthVoiceRegistry> registry = GetInstance();
|
||||
|
||||
return registry.forget();
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::Shutdown()
|
||||
{
|
||||
LOG(LogLevel::Debug, ("[%s] nsSynthVoiceRegistry::Shutdown()",
|
||||
(XRE_IsContentProcess()) ? "Content" : "Default"));
|
||||
gSynthVoiceRegistry = nullptr;
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::SendVoicesAndState(InfallibleTArray<RemoteVoice>* aVoices,
|
||||
InfallibleTArray<nsString>* aDefaults,
|
||||
bool* aIsSpeaking)
|
||||
{
|
||||
for (uint32_t i=0; i < mVoices.Length(); ++i) {
|
||||
RefPtr<VoiceData> voice = mVoices[i];
|
||||
|
||||
aVoices->AppendElement(RemoteVoice(voice->mUri, voice->mName, voice->mLang,
|
||||
voice->mIsLocal, voice->mIsQueued));
|
||||
}
|
||||
|
||||
for (uint32_t i=0; i < mDefaultVoices.Length(); ++i) {
|
||||
aDefaults->AppendElement(mDefaultVoices[i]->mUri);
|
||||
}
|
||||
|
||||
*aIsSpeaking = IsSpeaking();
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::RecvRemoveVoice(const nsAString& aUri)
|
||||
{
|
||||
// If we dont have a local instance of the registry yet, we will recieve current
|
||||
// voices at contruction time.
|
||||
if(!gSynthVoiceRegistry) {
|
||||
return;
|
||||
}
|
||||
|
||||
gSynthVoiceRegistry->RemoveVoice(nullptr, aUri);
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::RecvAddVoice(const RemoteVoice& aVoice)
|
||||
{
|
||||
// If we dont have a local instance of the registry yet, we will recieve current
|
||||
// voices at contruction time.
|
||||
if(!gSynthVoiceRegistry) {
|
||||
return;
|
||||
}
|
||||
|
||||
gSynthVoiceRegistry->AddVoiceImpl(nullptr, aVoice.voiceURI(),
|
||||
aVoice.name(), aVoice.lang(),
|
||||
aVoice.localService(), aVoice.queued());
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::RecvSetDefaultVoice(const nsAString& aUri, bool aIsDefault)
|
||||
{
|
||||
// If we dont have a local instance of the registry yet, we will recieve current
|
||||
// voices at contruction time.
|
||||
if(!gSynthVoiceRegistry) {
|
||||
return;
|
||||
}
|
||||
|
||||
gSynthVoiceRegistry->SetDefaultVoice(aUri, aIsDefault);
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::RecvIsSpeakingChanged(bool aIsSpeaking)
|
||||
{
|
||||
// If we dont have a local instance of the registry yet, we will get the
|
||||
// speaking state on construction.
|
||||
if(!gSynthVoiceRegistry) {
|
||||
return;
|
||||
}
|
||||
|
||||
gSynthVoiceRegistry->mIsSpeaking = aIsSpeaking;
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::RecvNotifyVoicesChanged()
|
||||
{
|
||||
// If we dont have a local instance of the registry yet, we don't care.
|
||||
if(!gSynthVoiceRegistry) {
|
||||
return;
|
||||
}
|
||||
|
||||
gSynthVoiceRegistry->NotifyVoicesChanged();
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSynthVoiceRegistry::AddVoice(nsISpeechService* aService,
|
||||
const nsAString& aUri,
|
||||
const nsAString& aName,
|
||||
const nsAString& aLang,
|
||||
bool aLocalService,
|
||||
bool aQueuesUtterances)
|
||||
{
|
||||
LOG(LogLevel::Debug,
|
||||
("nsSynthVoiceRegistry::AddVoice uri='%s' name='%s' lang='%s' local=%s queued=%s",
|
||||
NS_ConvertUTF16toUTF8(aUri).get(), NS_ConvertUTF16toUTF8(aName).get(),
|
||||
NS_ConvertUTF16toUTF8(aLang).get(),
|
||||
aLocalService ? "true" : "false",
|
||||
aQueuesUtterances ? "true" : "false"));
|
||||
|
||||
if(NS_WARN_IF(XRE_IsContentProcess())) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
return AddVoiceImpl(aService, aUri, aName, aLang, aLocalService, aQueuesUtterances);
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSynthVoiceRegistry::RemoveVoice(nsISpeechService* aService,
|
||||
const nsAString& aUri)
|
||||
{
|
||||
LOG(LogLevel::Debug,
|
||||
("nsSynthVoiceRegistry::RemoveVoice uri='%s' (%s)",
|
||||
NS_ConvertUTF16toUTF8(aUri).get(),
|
||||
(XRE_IsContentProcess()) ? "child" : "parent"));
|
||||
|
||||
bool found = false;
|
||||
VoiceData* retval = mUriVoiceMap.GetWeak(aUri, &found);
|
||||
|
||||
if(NS_WARN_IF(!(found))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
if(NS_WARN_IF(!(aService == retval->mService))) {
|
||||
return NS_ERROR_INVALID_ARG;
|
||||
}
|
||||
|
||||
mVoices.RemoveElement(retval);
|
||||
mDefaultVoices.RemoveElement(retval);
|
||||
mUriVoiceMap.Remove(aUri);
|
||||
|
||||
if (retval->mIsQueued && !MediaPrefs::WebSpeechForceGlobal()) {
|
||||
// Check if this is the last queued voice, and disable the global queue if
|
||||
// it is.
|
||||
bool queued = false;
|
||||
for (uint32_t i = 0; i < mVoices.Length(); i++) {
|
||||
VoiceData* voice = mVoices[i];
|
||||
if (voice->mIsQueued) {
|
||||
queued = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!queued) {
|
||||
mUseGlobalQueue = false;
|
||||
}
|
||||
}
|
||||
|
||||
nsTArray<SpeechSynthesisParent*> ssplist;
|
||||
GetAllSpeechSynthActors(ssplist);
|
||||
|
||||
for (uint32_t i = 0; i < ssplist.Length(); ++i)
|
||||
Unused << ssplist[i]->SendVoiceRemoved(nsString(aUri));
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSynthVoiceRegistry::NotifyVoicesChanged()
|
||||
{
|
||||
if (XRE_IsParentProcess()) {
|
||||
nsTArray<SpeechSynthesisParent*> ssplist;
|
||||
GetAllSpeechSynthActors(ssplist);
|
||||
|
||||
for (uint32_t i = 0; i < ssplist.Length(); ++i)
|
||||
Unused << ssplist[i]->SendNotifyVoicesChanged();
|
||||
}
|
||||
|
||||
nsCOMPtr<nsIObserverService> obs = mozilla::services::GetObserverService();
|
||||
if(NS_WARN_IF(!(obs))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
obs->NotifyObservers(nullptr, "synth-voices-changed", nullptr);
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSynthVoiceRegistry::SetDefaultVoice(const nsAString& aUri,
|
||||
bool aIsDefault)
|
||||
{
|
||||
bool found = false;
|
||||
VoiceData* retval = mUriVoiceMap.GetWeak(aUri, &found);
|
||||
if(NS_WARN_IF(!(found))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
mDefaultVoices.RemoveElement(retval);
|
||||
|
||||
LOG(LogLevel::Debug, ("nsSynthVoiceRegistry::SetDefaultVoice %s %s",
|
||||
NS_ConvertUTF16toUTF8(aUri).get(),
|
||||
aIsDefault ? "true" : "false"));
|
||||
|
||||
if (aIsDefault) {
|
||||
mDefaultVoices.AppendElement(retval);
|
||||
}
|
||||
|
||||
if (XRE_IsParentProcess()) {
|
||||
nsTArray<SpeechSynthesisParent*> ssplist;
|
||||
GetAllSpeechSynthActors(ssplist);
|
||||
|
||||
for (uint32_t i = 0; i < ssplist.Length(); ++i) {
|
||||
Unused << ssplist[i]->SendSetDefaultVoice(nsString(aUri), aIsDefault);
|
||||
}
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSynthVoiceRegistry::GetVoiceCount(uint32_t* aRetval)
|
||||
{
|
||||
*aRetval = mVoices.Length();
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSynthVoiceRegistry::GetVoice(uint32_t aIndex, nsAString& aRetval)
|
||||
{
|
||||
if(NS_WARN_IF(!(aIndex < mVoices.Length()))) {
|
||||
return NS_ERROR_INVALID_ARG;
|
||||
}
|
||||
|
||||
aRetval = mVoices[aIndex]->mUri;
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSynthVoiceRegistry::IsDefaultVoice(const nsAString& aUri, bool* aRetval)
|
||||
{
|
||||
bool found;
|
||||
VoiceData* voice = mUriVoiceMap.GetWeak(aUri, &found);
|
||||
if(NS_WARN_IF(!(found))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
for (int32_t i = mDefaultVoices.Length(); i > 0; ) {
|
||||
VoiceData* defaultVoice = mDefaultVoices[--i];
|
||||
|
||||
if (voice->mLang.Equals(defaultVoice->mLang)) {
|
||||
*aRetval = voice == defaultVoice;
|
||||
return NS_OK;
|
||||
}
|
||||
}
|
||||
|
||||
*aRetval = false;
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSynthVoiceRegistry::IsLocalVoice(const nsAString& aUri, bool* aRetval)
|
||||
{
|
||||
bool found;
|
||||
VoiceData* voice = mUriVoiceMap.GetWeak(aUri, &found);
|
||||
if(NS_WARN_IF(!(found))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
*aRetval = voice->mIsLocal;
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSynthVoiceRegistry::GetVoiceLang(const nsAString& aUri, nsAString& aRetval)
|
||||
{
|
||||
bool found;
|
||||
VoiceData* voice = mUriVoiceMap.GetWeak(aUri, &found);
|
||||
if(NS_WARN_IF(!(found))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
aRetval = voice->mLang;
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsSynthVoiceRegistry::GetVoiceName(const nsAString& aUri, nsAString& aRetval)
|
||||
{
|
||||
bool found;
|
||||
VoiceData* voice = mUriVoiceMap.GetWeak(aUri, &found);
|
||||
if(NS_WARN_IF(!(found))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
aRetval = voice->mName;
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
nsresult
|
||||
nsSynthVoiceRegistry::AddVoiceImpl(nsISpeechService* aService,
|
||||
const nsAString& aUri,
|
||||
const nsAString& aName,
|
||||
const nsAString& aLang,
|
||||
bool aLocalService,
|
||||
bool aQueuesUtterances)
|
||||
{
|
||||
bool found = false;
|
||||
mUriVoiceMap.GetWeak(aUri, &found);
|
||||
if(NS_WARN_IF(found)) {
|
||||
return NS_ERROR_INVALID_ARG;
|
||||
}
|
||||
|
||||
RefPtr<VoiceData> voice = new VoiceData(aService, aUri, aName, aLang,
|
||||
aLocalService, aQueuesUtterances);
|
||||
|
||||
mVoices.AppendElement(voice);
|
||||
mUriVoiceMap.Put(aUri, voice);
|
||||
mUseGlobalQueue |= aQueuesUtterances;
|
||||
|
||||
nsTArray<SpeechSynthesisParent*> ssplist;
|
||||
GetAllSpeechSynthActors(ssplist);
|
||||
|
||||
if (!ssplist.IsEmpty()) {
|
||||
mozilla::dom::RemoteVoice ssvoice(nsString(aUri),
|
||||
nsString(aName),
|
||||
nsString(aLang),
|
||||
aLocalService,
|
||||
aQueuesUtterances);
|
||||
|
||||
for (uint32_t i = 0; i < ssplist.Length(); ++i) {
|
||||
Unused << ssplist[i]->SendVoiceAdded(ssvoice);
|
||||
}
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
bool
|
||||
nsSynthVoiceRegistry::FindVoiceByLang(const nsAString& aLang,
|
||||
VoiceData** aRetval)
|
||||
{
|
||||
nsAString::const_iterator dashPos, start, end;
|
||||
aLang.BeginReading(start);
|
||||
aLang.EndReading(end);
|
||||
|
||||
while (true) {
|
||||
nsAutoString langPrefix(Substring(start, end));
|
||||
|
||||
for (int32_t i = mDefaultVoices.Length(); i > 0; ) {
|
||||
VoiceData* voice = mDefaultVoices[--i];
|
||||
|
||||
if (StringBeginsWith(voice->mLang, langPrefix)) {
|
||||
*aRetval = voice;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
for (int32_t i = mVoices.Length(); i > 0; ) {
|
||||
VoiceData* voice = mVoices[--i];
|
||||
|
||||
if (StringBeginsWith(voice->mLang, langPrefix)) {
|
||||
*aRetval = voice;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
dashPos = end;
|
||||
end = start;
|
||||
|
||||
if (!RFindInReadable(NS_LITERAL_STRING("-"), end, dashPos)) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
VoiceData*
|
||||
nsSynthVoiceRegistry::FindBestMatch(const nsAString& aUri,
|
||||
const nsAString& aLang)
|
||||
{
|
||||
if (mVoices.IsEmpty()) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
bool found = false;
|
||||
VoiceData* retval = mUriVoiceMap.GetWeak(aUri, &found);
|
||||
|
||||
if (found) {
|
||||
LOG(LogLevel::Debug, ("nsSynthVoiceRegistry::FindBestMatch - Matched URI"));
|
||||
return retval;
|
||||
}
|
||||
|
||||
// Try finding a match for given voice.
|
||||
if (!aLang.IsVoid() && !aLang.IsEmpty()) {
|
||||
if (FindVoiceByLang(aLang, &retval)) {
|
||||
LOG(LogLevel::Debug,
|
||||
("nsSynthVoiceRegistry::FindBestMatch - Matched language (%s ~= %s)",
|
||||
NS_ConvertUTF16toUTF8(aLang).get(),
|
||||
NS_ConvertUTF16toUTF8(retval->mLang).get()));
|
||||
|
||||
return retval;
|
||||
}
|
||||
}
|
||||
|
||||
// Try UI language.
|
||||
nsresult rv;
|
||||
nsCOMPtr<nsILocaleService> localeService = do_GetService(NS_LOCALESERVICE_CONTRACTID, &rv);
|
||||
if (NS_WARN_IF(NS_FAILED(rv))) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
nsAutoString uiLang;
|
||||
rv = localeService->GetLocaleComponentForUserAgent(uiLang);
|
||||
if (NS_WARN_IF(NS_FAILED(rv))) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
if (FindVoiceByLang(uiLang, &retval)) {
|
||||
LOG(LogLevel::Debug,
|
||||
("nsSynthVoiceRegistry::FindBestMatch - Matched UI language (%s ~= %s)",
|
||||
NS_ConvertUTF16toUTF8(uiLang).get(),
|
||||
NS_ConvertUTF16toUTF8(retval->mLang).get()));
|
||||
|
||||
return retval;
|
||||
}
|
||||
|
||||
// Try en-US, the language of locale "C"
|
||||
if (FindVoiceByLang(NS_LITERAL_STRING("en-US"), &retval)) {
|
||||
LOG(LogLevel::Debug,
|
||||
("nsSynthVoiceRegistry::FindBestMatch - Matched C locale language (en-US ~= %s)",
|
||||
NS_ConvertUTF16toUTF8(retval->mLang).get()));
|
||||
|
||||
return retval;
|
||||
}
|
||||
|
||||
// The top default voice is better than nothing...
|
||||
if (!mDefaultVoices.IsEmpty()) {
|
||||
return mDefaultVoices.LastElement();
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
already_AddRefed<nsSpeechTask>
|
||||
nsSynthVoiceRegistry::SpeakUtterance(SpeechSynthesisUtterance& aUtterance,
|
||||
const nsAString& aDocLang)
|
||||
{
|
||||
nsString lang = nsString(aUtterance.mLang.IsEmpty() ? aDocLang : aUtterance.mLang);
|
||||
nsAutoString uri;
|
||||
|
||||
if (aUtterance.mVoice) {
|
||||
aUtterance.mVoice->GetVoiceURI(uri);
|
||||
}
|
||||
|
||||
// Get current audio volume to apply speech call
|
||||
float volume = aUtterance.Volume();
|
||||
RefPtr<AudioChannelService> service = AudioChannelService::GetOrCreate();
|
||||
if (service) {
|
||||
if (nsCOMPtr<nsPIDOMWindowInner> topWindow = aUtterance.GetOwner()) {
|
||||
// TODO : use audio channel agent, open new bug to fix it.
|
||||
uint32_t channel = static_cast<uint32_t>(AudioChannelService::GetDefaultAudioChannel());
|
||||
AudioPlaybackConfig config = service->GetMediaConfig(topWindow->GetOuterWindow(),
|
||||
channel);
|
||||
volume = config.mMuted ? 0.0f : config.mVolume * volume;
|
||||
}
|
||||
}
|
||||
|
||||
RefPtr<nsSpeechTask> task;
|
||||
if (XRE_IsContentProcess()) {
|
||||
task = new SpeechTaskChild(&aUtterance);
|
||||
SpeechSynthesisRequestChild* actor =
|
||||
new SpeechSynthesisRequestChild(static_cast<SpeechTaskChild*>(task.get()));
|
||||
mSpeechSynthChild->SendPSpeechSynthesisRequestConstructor(actor,
|
||||
aUtterance.mText,
|
||||
lang,
|
||||
uri,
|
||||
volume,
|
||||
aUtterance.Rate(),
|
||||
aUtterance.Pitch());
|
||||
} else {
|
||||
task = new nsSpeechTask(&aUtterance);
|
||||
Speak(aUtterance.mText, lang, uri,
|
||||
volume, aUtterance.Rate(), aUtterance.Pitch(), task);
|
||||
}
|
||||
|
||||
return task.forget();
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::Speak(const nsAString& aText,
|
||||
const nsAString& aLang,
|
||||
const nsAString& aUri,
|
||||
const float& aVolume,
|
||||
const float& aRate,
|
||||
const float& aPitch,
|
||||
nsSpeechTask* aTask)
|
||||
{
|
||||
MOZ_ASSERT(XRE_IsParentProcess());
|
||||
|
||||
VoiceData* voice = FindBestMatch(aUri, aLang);
|
||||
|
||||
if (!voice) {
|
||||
NS_WARNING("No voices found.");
|
||||
aTask->DispatchError(0, 0);
|
||||
return;
|
||||
}
|
||||
|
||||
aTask->SetChosenVoiceURI(voice->mUri);
|
||||
|
||||
if (mUseGlobalQueue || MediaPrefs::WebSpeechForceGlobal()) {
|
||||
LOG(LogLevel::Debug,
|
||||
("nsSynthVoiceRegistry::Speak queueing text='%s' lang='%s' uri='%s' rate=%f pitch=%f",
|
||||
NS_ConvertUTF16toUTF8(aText).get(), NS_ConvertUTF16toUTF8(aLang).get(),
|
||||
NS_ConvertUTF16toUTF8(aUri).get(), aRate, aPitch));
|
||||
RefPtr<GlobalQueueItem> item = new GlobalQueueItem(voice, aTask, aText,
|
||||
aVolume, aRate, aPitch);
|
||||
mGlobalQueue.AppendElement(item);
|
||||
|
||||
if (mGlobalQueue.Length() == 1) {
|
||||
SpeakImpl(item->mVoice, item->mTask, item->mText, item->mVolume, item->mRate,
|
||||
item->mPitch);
|
||||
}
|
||||
} else {
|
||||
SpeakImpl(voice, aTask, aText, aVolume, aRate, aPitch);
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::SpeakNext()
|
||||
{
|
||||
MOZ_ASSERT(XRE_IsParentProcess());
|
||||
|
||||
LOG(LogLevel::Debug,
|
||||
("nsSynthVoiceRegistry::SpeakNext %d", mGlobalQueue.IsEmpty()));
|
||||
|
||||
SetIsSpeaking(false);
|
||||
|
||||
if (mGlobalQueue.IsEmpty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
mGlobalQueue.RemoveElementAt(0);
|
||||
|
||||
while (!mGlobalQueue.IsEmpty()) {
|
||||
RefPtr<GlobalQueueItem> item = mGlobalQueue.ElementAt(0);
|
||||
if (item->mTask->IsPreCanceled()) {
|
||||
mGlobalQueue.RemoveElementAt(0);
|
||||
continue;
|
||||
}
|
||||
if (!item->mTask->IsPrePaused()) {
|
||||
SpeakImpl(item->mVoice, item->mTask, item->mText, item->mVolume,
|
||||
item->mRate, item->mPitch);
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::ResumeQueue()
|
||||
{
|
||||
MOZ_ASSERT(XRE_IsParentProcess());
|
||||
LOG(LogLevel::Debug,
|
||||
("nsSynthVoiceRegistry::ResumeQueue %d", mGlobalQueue.IsEmpty()));
|
||||
|
||||
if (mGlobalQueue.IsEmpty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
RefPtr<GlobalQueueItem> item = mGlobalQueue.ElementAt(0);
|
||||
if (!item->mTask->IsPrePaused()) {
|
||||
SpeakImpl(item->mVoice, item->mTask, item->mText, item->mVolume,
|
||||
item->mRate, item->mPitch);
|
||||
}
|
||||
}
|
||||
|
||||
bool
|
||||
nsSynthVoiceRegistry::IsSpeaking()
|
||||
{
|
||||
return mIsSpeaking;
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::SetIsSpeaking(bool aIsSpeaking)
|
||||
{
|
||||
MOZ_ASSERT(XRE_IsParentProcess());
|
||||
|
||||
// Only set to 'true' if global queue is enabled.
|
||||
mIsSpeaking =
|
||||
aIsSpeaking && (mUseGlobalQueue || MediaPrefs::WebSpeechForceGlobal());
|
||||
|
||||
nsTArray<SpeechSynthesisParent*> ssplist;
|
||||
GetAllSpeechSynthActors(ssplist);
|
||||
for (uint32_t i = 0; i < ssplist.Length(); ++i) {
|
||||
Unused << ssplist[i]->SendIsSpeakingChanged(aIsSpeaking);
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
nsSynthVoiceRegistry::SpeakImpl(VoiceData* aVoice,
|
||||
nsSpeechTask* aTask,
|
||||
const nsAString& aText,
|
||||
const float& aVolume,
|
||||
const float& aRate,
|
||||
const float& aPitch)
|
||||
{
|
||||
LOG(LogLevel::Debug,
|
||||
("nsSynthVoiceRegistry::SpeakImpl queueing text='%s' uri='%s' rate=%f pitch=%f",
|
||||
NS_ConvertUTF16toUTF8(aText).get(), NS_ConvertUTF16toUTF8(aVoice->mUri).get(),
|
||||
aRate, aPitch));
|
||||
|
||||
SpeechServiceType serviceType;
|
||||
|
||||
DebugOnly<nsresult> rv = aVoice->mService->GetServiceType(&serviceType);
|
||||
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv), "Failed to get speech service type");
|
||||
|
||||
if (serviceType == nsISpeechService::SERVICETYPE_INDIRECT_AUDIO) {
|
||||
aTask->InitIndirectAudio();
|
||||
} else {
|
||||
aTask->InitDirectAudio();
|
||||
}
|
||||
|
||||
if (NS_FAILED(aVoice->mService->Speak(aText, aVoice->mUri, aVolume, aRate,
|
||||
aPitch, aTask))) {
|
||||
if (serviceType == nsISpeechService::SERVICETYPE_INDIRECT_AUDIO) {
|
||||
aTask->DispatchError(0, 0);
|
||||
}
|
||||
// XXX When using direct audio, no way to dispatch error
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
109
dom/media/webspeech/synth/nsSynthVoiceRegistry.h
Normal file
109
dom/media/webspeech/synth/nsSynthVoiceRegistry.h
Normal file
|
|
@ -0,0 +1,109 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_nsSynthVoiceRegistry_h
|
||||
#define mozilla_dom_nsSynthVoiceRegistry_h
|
||||
|
||||
#include "nsISynthVoiceRegistry.h"
|
||||
#include "nsRefPtrHashtable.h"
|
||||
#include "nsTArray.h"
|
||||
#include "MediaStreamGraph.h"
|
||||
|
||||
class nsISpeechService;
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class RemoteVoice;
|
||||
class SpeechSynthesisUtterance;
|
||||
class SpeechSynthesisChild;
|
||||
class nsSpeechTask;
|
||||
class VoiceData;
|
||||
class GlobalQueueItem;
|
||||
|
||||
class nsSynthVoiceRegistry final : public nsISynthVoiceRegistry
|
||||
{
|
||||
public:
|
||||
NS_DECL_ISUPPORTS
|
||||
NS_DECL_NSISYNTHVOICEREGISTRY
|
||||
|
||||
nsSynthVoiceRegistry();
|
||||
|
||||
already_AddRefed<nsSpeechTask> SpeakUtterance(SpeechSynthesisUtterance& aUtterance,
|
||||
const nsAString& aDocLang);
|
||||
|
||||
void Speak(const nsAString& aText, const nsAString& aLang,
|
||||
const nsAString& aUri, const float& aVolume, const float& aRate,
|
||||
const float& aPitch, nsSpeechTask* aTask);
|
||||
|
||||
void SendVoicesAndState(InfallibleTArray<RemoteVoice>* aVoices,
|
||||
InfallibleTArray<nsString>* aDefaults,
|
||||
bool* aIsSpeaking);
|
||||
|
||||
void SpeakNext();
|
||||
|
||||
void ResumeQueue();
|
||||
|
||||
bool IsSpeaking();
|
||||
|
||||
void SetIsSpeaking(bool aIsSpeaking);
|
||||
|
||||
static nsSynthVoiceRegistry* GetInstance();
|
||||
|
||||
static already_AddRefed<nsSynthVoiceRegistry> GetInstanceForService();
|
||||
|
||||
static void RecvRemoveVoice(const nsAString& aUri);
|
||||
|
||||
static void RecvAddVoice(const RemoteVoice& aVoice);
|
||||
|
||||
static void RecvSetDefaultVoice(const nsAString& aUri, bool aIsDefault);
|
||||
|
||||
static void RecvIsSpeakingChanged(bool aIsSpeaking);
|
||||
|
||||
static void RecvNotifyVoicesChanged();
|
||||
|
||||
static void Shutdown();
|
||||
|
||||
private:
|
||||
virtual ~nsSynthVoiceRegistry();
|
||||
|
||||
VoiceData* FindBestMatch(const nsAString& aUri, const nsAString& lang);
|
||||
|
||||
bool FindVoiceByLang(const nsAString& aLang, VoiceData** aRetval);
|
||||
|
||||
nsresult AddVoiceImpl(nsISpeechService* aService,
|
||||
const nsAString& aUri,
|
||||
const nsAString& aName,
|
||||
const nsAString& aLang,
|
||||
bool aLocalService,
|
||||
bool aQueuesUtterances);
|
||||
|
||||
void SpeakImpl(VoiceData* aVoice,
|
||||
nsSpeechTask* aTask,
|
||||
const nsAString& aText,
|
||||
const float& aVolume,
|
||||
const float& aRate,
|
||||
const float& aPitch);
|
||||
|
||||
nsTArray<RefPtr<VoiceData>> mVoices;
|
||||
|
||||
nsTArray<RefPtr<VoiceData>> mDefaultVoices;
|
||||
|
||||
nsRefPtrHashtable<nsStringHashKey, VoiceData> mUriVoiceMap;
|
||||
|
||||
SpeechSynthesisChild* mSpeechSynthChild;
|
||||
|
||||
bool mUseGlobalQueue;
|
||||
|
||||
nsTArray<RefPtr<GlobalQueueItem>> mGlobalQueue;
|
||||
|
||||
bool mIsSpeaking;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
58
dom/media/webspeech/synth/pico/PicoModule.cpp
Normal file
58
dom/media/webspeech/synth/pico/PicoModule.cpp
Normal file
|
|
@ -0,0 +1,58 @@
|
|||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "mozilla/ModuleUtils.h"
|
||||
#include "nsIClassInfoImpl.h"
|
||||
|
||||
#ifdef MOZ_WEBRTC
|
||||
|
||||
#include "nsPicoService.h"
|
||||
|
||||
using namespace mozilla::dom;
|
||||
|
||||
#define PICOSERVICE_CID \
|
||||
{0x346c4fc8, 0x12fe, 0x459c, {0x81, 0x19, 0x9a, 0xa7, 0x73, 0x37, 0x7f, 0xf4}}
|
||||
|
||||
#define PICOSERVICE_CONTRACTID "@mozilla.org/synthpico;1"
|
||||
|
||||
// Defines nsPicoServiceConstructor
|
||||
NS_GENERIC_FACTORY_SINGLETON_CONSTRUCTOR(nsPicoService,
|
||||
nsPicoService::GetInstanceForService)
|
||||
|
||||
// Defines kPICOSERVICE_CID
|
||||
NS_DEFINE_NAMED_CID(PICOSERVICE_CID);
|
||||
|
||||
static const mozilla::Module::CIDEntry kCIDs[] = {
|
||||
{ &kPICOSERVICE_CID, true, nullptr, nsPicoServiceConstructor },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static const mozilla::Module::ContractIDEntry kContracts[] = {
|
||||
{ PICOSERVICE_CONTRACTID, &kPICOSERVICE_CID },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static const mozilla::Module::CategoryEntry kCategories[] = {
|
||||
{ "profile-after-change", "Pico Speech Synth", PICOSERVICE_CONTRACTID },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static void
|
||||
UnloadPicoModule()
|
||||
{
|
||||
nsPicoService::Shutdown();
|
||||
}
|
||||
|
||||
static const mozilla::Module kModule = {
|
||||
mozilla::Module::kVersion,
|
||||
kCIDs,
|
||||
kContracts,
|
||||
kCategories,
|
||||
nullptr,
|
||||
nullptr,
|
||||
UnloadPicoModule
|
||||
};
|
||||
|
||||
NSMODULE_DEFN(synthpico) = &kModule;
|
||||
#endif
|
||||
13
dom/media/webspeech/synth/pico/moz.build
Normal file
13
dom/media/webspeech/synth/pico/moz.build
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
# -*- Mode: python; indent-tabs-mode: nil; tab-width: 40 -*-
|
||||
# vim: set filetype=python:
|
||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
UNIFIED_SOURCES += [
|
||||
'nsPicoService.cpp',
|
||||
'PicoModule.cpp'
|
||||
]
|
||||
include('/ipc/chromium/chromium-config.mozbuild')
|
||||
|
||||
FINAL_LIBRARY = 'xul'
|
||||
761
dom/media/webspeech/synth/pico/nsPicoService.cpp
Normal file
761
dom/media/webspeech/synth/pico/nsPicoService.cpp
Normal file
|
|
@ -0,0 +1,761 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsISupports.h"
|
||||
#include "nsPicoService.h"
|
||||
#include "nsPrintfCString.h"
|
||||
#include "nsIWeakReferenceUtils.h"
|
||||
#include "SharedBuffer.h"
|
||||
#include "nsISimpleEnumerator.h"
|
||||
|
||||
#include "mozilla/dom/nsSynthVoiceRegistry.h"
|
||||
#include "mozilla/dom/nsSpeechTask.h"
|
||||
|
||||
#include "nsIFile.h"
|
||||
#include "nsThreadUtils.h"
|
||||
#include "prenv.h"
|
||||
#include "mozilla/Preferences.h"
|
||||
#include "mozilla/DebugOnly.h"
|
||||
#include <dlfcn.h>
|
||||
|
||||
// Pico API constants
|
||||
|
||||
// Size of memory allocated for pico engine and voice resources.
|
||||
// We only have one voice and its resources loaded at once, so this
|
||||
// should always be enough.
|
||||
#define PICO_MEM_SIZE 2500000
|
||||
|
||||
// Max length of returned strings. Pico will never return longer strings,
|
||||
// so this amount should be good enough for preallocating.
|
||||
#define PICO_RETSTRINGSIZE 200
|
||||
|
||||
// Max amount we want from a single call of pico_getData
|
||||
#define PICO_MAX_CHUNK_SIZE 128
|
||||
|
||||
// Arbitrary name for loaded voice, it doesn't mean anything outside of Pico
|
||||
#define PICO_VOICE_NAME "pico"
|
||||
|
||||
// Return status from pico_getData meaning there is more data in the pipeline
|
||||
// to get from more calls to pico_getData
|
||||
#define PICO_STEP_BUSY 201
|
||||
|
||||
// For performing a "soft" reset between utterances. This is used when one
|
||||
// utterance is interrupted by a new one.
|
||||
#define PICO_RESET_SOFT 0x10
|
||||
|
||||
// Currently, Pico only provides mono output.
|
||||
#define PICO_CHANNELS_NUM 1
|
||||
|
||||
// Pico's sample rate is always 16000
|
||||
#define PICO_SAMPLE_RATE 16000
|
||||
|
||||
// The path to the language files in Gonk
|
||||
#define GONK_PICO_LANG_PATH "/system/tts/lang_pico"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
StaticRefPtr<nsPicoService> nsPicoService::sSingleton;
|
||||
|
||||
class PicoApi
|
||||
{
|
||||
public:
|
||||
|
||||
PicoApi() : mInitialized(false) {}
|
||||
|
||||
bool Init()
|
||||
{
|
||||
if (mInitialized) {
|
||||
return true;
|
||||
}
|
||||
|
||||
void* handle = dlopen("libttspico.so", RTLD_LAZY);
|
||||
|
||||
if (!handle) {
|
||||
NS_WARNING("Failed to open libttspico.so, pico cannot run");
|
||||
return false;
|
||||
}
|
||||
|
||||
pico_initialize =
|
||||
(pico_Status (*)(void*, uint32_t, pico_System*))dlsym(
|
||||
handle, "pico_initialize");
|
||||
|
||||
pico_terminate =
|
||||
(pico_Status (*)(pico_System*))dlsym(handle, "pico_terminate");
|
||||
|
||||
pico_getSystemStatusMessage =
|
||||
(pico_Status (*)(pico_System, pico_Status, pico_Retstring))dlsym(
|
||||
handle, "pico_getSystemStatusMessage");;
|
||||
|
||||
pico_loadResource =
|
||||
(pico_Status (*)(pico_System, const char*, pico_Resource*))dlsym(
|
||||
handle, "pico_loadResource");
|
||||
|
||||
pico_unloadResource =
|
||||
(pico_Status (*)(pico_System, pico_Resource*))dlsym(
|
||||
handle, "pico_unloadResource");
|
||||
|
||||
pico_getResourceName =
|
||||
(pico_Status (*)(pico_System, pico_Resource, pico_Retstring))dlsym(
|
||||
handle, "pico_getResourceName");
|
||||
|
||||
pico_createVoiceDefinition =
|
||||
(pico_Status (*)(pico_System, const char*))dlsym(
|
||||
handle, "pico_createVoiceDefinition");
|
||||
|
||||
pico_addResourceToVoiceDefinition =
|
||||
(pico_Status (*)(pico_System, const char*, const char*))dlsym(
|
||||
handle, "pico_addResourceToVoiceDefinition");
|
||||
|
||||
pico_releaseVoiceDefinition =
|
||||
(pico_Status (*)(pico_System, const char*))dlsym(
|
||||
handle, "pico_releaseVoiceDefinition");
|
||||
|
||||
pico_newEngine =
|
||||
(pico_Status (*)(pico_System, const char*, pico_Engine*))dlsym(
|
||||
handle, "pico_newEngine");
|
||||
|
||||
pico_disposeEngine =
|
||||
(pico_Status (*)(pico_System, pico_Engine*))dlsym(
|
||||
handle, "pico_disposeEngine");
|
||||
|
||||
pico_resetEngine =
|
||||
(pico_Status (*)(pico_Engine, int32_t))dlsym(handle, "pico_resetEngine");
|
||||
|
||||
pico_putTextUtf8 =
|
||||
(pico_Status (*)(pico_Engine, const char*, const int16_t, int16_t*))dlsym(
|
||||
handle, "pico_putTextUtf8");
|
||||
|
||||
pico_getData =
|
||||
(pico_Status (*)(pico_Engine, void*, int16_t, int16_t*, int16_t*))dlsym(
|
||||
handle, "pico_getData");
|
||||
|
||||
mInitialized = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
typedef signed int pico_Status;
|
||||
typedef char pico_Retstring[PICO_RETSTRINGSIZE];
|
||||
|
||||
pico_Status (* pico_initialize)(void*, uint32_t, pico_System*);
|
||||
pico_Status (* pico_terminate)(pico_System*);
|
||||
pico_Status (* pico_getSystemStatusMessage)(
|
||||
pico_System, pico_Status, pico_Retstring);
|
||||
|
||||
pico_Status (* pico_loadResource)(pico_System, const char*, pico_Resource*);
|
||||
pico_Status (* pico_unloadResource)(pico_System, pico_Resource*);
|
||||
pico_Status (* pico_getResourceName)(
|
||||
pico_System, pico_Resource, pico_Retstring);
|
||||
pico_Status (* pico_createVoiceDefinition)(pico_System, const char*);
|
||||
pico_Status (* pico_addResourceToVoiceDefinition)(
|
||||
pico_System, const char*, const char*);
|
||||
pico_Status (* pico_releaseVoiceDefinition)(pico_System, const char*);
|
||||
pico_Status (* pico_newEngine)(pico_System, const char*, pico_Engine*);
|
||||
pico_Status (* pico_disposeEngine)(pico_System, pico_Engine*);
|
||||
|
||||
pico_Status (* pico_resetEngine)(pico_Engine, int32_t);
|
||||
pico_Status (* pico_putTextUtf8)(
|
||||
pico_Engine, const char*, const int16_t, int16_t*);
|
||||
pico_Status (* pico_getData)(
|
||||
pico_Engine, void*, const int16_t, int16_t*, int16_t*);
|
||||
|
||||
private:
|
||||
|
||||
bool mInitialized;
|
||||
|
||||
} sPicoApi;
|
||||
|
||||
#define PICO_ENSURE_SUCCESS_VOID(_funcName, _status) \
|
||||
if (_status < 0) { \
|
||||
PicoApi::pico_Retstring message; \
|
||||
sPicoApi.pico_getSystemStatusMessage( \
|
||||
nsPicoService::sSingleton->mPicoSystem, _status, message); \
|
||||
NS_WARNING( \
|
||||
nsPrintfCString("Error running %s: %s", _funcName, message).get()); \
|
||||
return; \
|
||||
}
|
||||
|
||||
#define PICO_ENSURE_SUCCESS(_funcName, _status, _rv) \
|
||||
if (_status < 0) { \
|
||||
PicoApi::pico_Retstring message; \
|
||||
sPicoApi.pico_getSystemStatusMessage( \
|
||||
nsPicoService::sSingleton->mPicoSystem, _status, message); \
|
||||
NS_WARNING( \
|
||||
nsPrintfCString("Error running %s: %s", _funcName, message).get()); \
|
||||
return _rv; \
|
||||
}
|
||||
|
||||
class PicoVoice
|
||||
{
|
||||
public:
|
||||
|
||||
PicoVoice(const nsAString& aLanguage)
|
||||
: mLanguage(aLanguage) {}
|
||||
|
||||
NS_INLINE_DECL_THREADSAFE_REFCOUNTING(PicoVoice)
|
||||
|
||||
// Voice language, in BCB-47 syntax
|
||||
nsString mLanguage;
|
||||
|
||||
// Language resource file
|
||||
nsCString mTaFile;
|
||||
|
||||
// Speaker resource file
|
||||
nsCString mSgFile;
|
||||
|
||||
private:
|
||||
~PicoVoice() {}
|
||||
};
|
||||
|
||||
class PicoCallbackRunnable : public Runnable,
|
||||
public nsISpeechTaskCallback
|
||||
{
|
||||
friend class PicoSynthDataRunnable;
|
||||
|
||||
public:
|
||||
PicoCallbackRunnable(const nsAString& aText, PicoVoice* aVoice,
|
||||
float aRate, float aPitch, nsISpeechTask* aTask,
|
||||
nsPicoService* aService)
|
||||
: mText(NS_ConvertUTF16toUTF8(aText))
|
||||
, mRate(aRate)
|
||||
, mPitch(aPitch)
|
||||
, mFirstData(true)
|
||||
, mTask(aTask)
|
||||
, mVoice(aVoice)
|
||||
, mService(aService) { }
|
||||
|
||||
NS_DECL_ISUPPORTS_INHERITED
|
||||
NS_DECL_NSISPEECHTASKCALLBACK
|
||||
|
||||
NS_IMETHOD Run() override;
|
||||
|
||||
bool IsCurrentTask() { return mService->mCurrentTask == mTask; }
|
||||
|
||||
private:
|
||||
~PicoCallbackRunnable() { }
|
||||
|
||||
void DispatchSynthDataRunnable(already_AddRefed<SharedBuffer>&& aBuffer,
|
||||
size_t aBufferSize);
|
||||
|
||||
nsCString mText;
|
||||
|
||||
float mRate;
|
||||
|
||||
float mPitch;
|
||||
|
||||
bool mFirstData;
|
||||
|
||||
// We use this pointer to compare it with the current service task.
|
||||
// If they differ, this runnable should stop.
|
||||
nsISpeechTask* mTask;
|
||||
|
||||
// We hold a strong reference to the service, which in turn holds
|
||||
// a strong reference to this voice.
|
||||
PicoVoice* mVoice;
|
||||
|
||||
// By holding a strong reference to the service we guarantee that it won't be
|
||||
// destroyed before this runnable.
|
||||
RefPtr<nsPicoService> mService;
|
||||
};
|
||||
|
||||
NS_IMPL_ISUPPORTS_INHERITED(PicoCallbackRunnable, Runnable, nsISpeechTaskCallback)
|
||||
|
||||
// Runnable
|
||||
|
||||
NS_IMETHODIMP
|
||||
PicoCallbackRunnable::Run()
|
||||
{
|
||||
MOZ_ASSERT(!NS_IsMainThread());
|
||||
PicoApi::pico_Status status = 0;
|
||||
|
||||
if (mService->CurrentVoice() != mVoice) {
|
||||
mService->LoadEngine(mVoice);
|
||||
} else {
|
||||
status = sPicoApi.pico_resetEngine(mService->mPicoEngine, PICO_RESET_SOFT);
|
||||
PICO_ENSURE_SUCCESS("pico_unloadResource", status, NS_ERROR_FAILURE);
|
||||
}
|
||||
|
||||
// Add SSML markup for pitch and rate. Pico uses a minimal parser,
|
||||
// so no namespace is needed.
|
||||
nsPrintfCString markedUpText(
|
||||
"<pitch level=\"%0.0f\"><speed level=\"%0.0f\">%s</speed></pitch>",
|
||||
std::min(std::max(50.0f, mPitch * 100), 200.0f),
|
||||
std::min(std::max(20.0f, mRate * 100), 500.0f),
|
||||
mText.get());
|
||||
|
||||
const char* text = markedUpText.get();
|
||||
size_t buffer_size = 512, buffer_offset = 0;
|
||||
RefPtr<SharedBuffer> buffer = SharedBuffer::Create(buffer_size);
|
||||
int16_t text_offset = 0, bytes_recv = 0, bytes_sent = 0, out_data_type = 0;
|
||||
int16_t text_remaining = markedUpText.Length() + 1;
|
||||
|
||||
// Run this loop while this is the current task
|
||||
while (IsCurrentTask()) {
|
||||
if (text_remaining) {
|
||||
status = sPicoApi.pico_putTextUtf8(mService->mPicoEngine,
|
||||
text + text_offset, text_remaining,
|
||||
&bytes_sent);
|
||||
PICO_ENSURE_SUCCESS("pico_putTextUtf8", status, NS_ERROR_FAILURE);
|
||||
// XXX: End speech task on error
|
||||
text_remaining -= bytes_sent;
|
||||
text_offset += bytes_sent;
|
||||
} else {
|
||||
// If we already fed all the text to the engine, send a zero length buffer
|
||||
// and quit.
|
||||
DispatchSynthDataRunnable(already_AddRefed<SharedBuffer>(), 0);
|
||||
break;
|
||||
}
|
||||
|
||||
do {
|
||||
// Run this loop while the result of getData is STEP_BUSY, when it finishes
|
||||
// synthesizing audio for the given text, it returns STEP_IDLE. We then
|
||||
// break to the outer loop and feed more text, if there is any left.
|
||||
if (!IsCurrentTask()) {
|
||||
// If the task has changed, quit.
|
||||
break;
|
||||
}
|
||||
|
||||
if (buffer_size - buffer_offset < PICO_MAX_CHUNK_SIZE) {
|
||||
// The next audio chunk retrieved may be bigger than our buffer,
|
||||
// so send the data and flush the buffer.
|
||||
DispatchSynthDataRunnable(buffer.forget(), buffer_offset);
|
||||
buffer_offset = 0;
|
||||
buffer = SharedBuffer::Create(buffer_size);
|
||||
}
|
||||
|
||||
status = sPicoApi.pico_getData(mService->mPicoEngine,
|
||||
(uint8_t*)buffer->Data() + buffer_offset,
|
||||
PICO_MAX_CHUNK_SIZE,
|
||||
&bytes_recv, &out_data_type);
|
||||
PICO_ENSURE_SUCCESS("pico_getData", status, NS_ERROR_FAILURE);
|
||||
buffer_offset += bytes_recv;
|
||||
} while (status == PICO_STEP_BUSY);
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
void
|
||||
PicoCallbackRunnable::DispatchSynthDataRunnable(
|
||||
already_AddRefed<SharedBuffer>&& aBuffer, size_t aBufferSize)
|
||||
{
|
||||
class PicoSynthDataRunnable final : public Runnable
|
||||
{
|
||||
public:
|
||||
PicoSynthDataRunnable(already_AddRefed<SharedBuffer>& aBuffer,
|
||||
size_t aBufferSize, bool aFirstData,
|
||||
PicoCallbackRunnable* aCallback)
|
||||
: mBuffer(aBuffer)
|
||||
, mBufferSize(aBufferSize)
|
||||
, mFirstData(aFirstData)
|
||||
, mCallback(aCallback) {
|
||||
}
|
||||
|
||||
NS_IMETHOD Run() override
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
|
||||
if (!mCallback->IsCurrentTask()) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
nsISpeechTask* task = mCallback->mTask;
|
||||
|
||||
if (mFirstData) {
|
||||
task->Setup(mCallback, PICO_CHANNELS_NUM, PICO_SAMPLE_RATE, 2);
|
||||
}
|
||||
|
||||
return task->SendAudioNative(
|
||||
mBufferSize ? static_cast<short*>(mBuffer->Data()) : nullptr, mBufferSize / 2);
|
||||
}
|
||||
|
||||
private:
|
||||
RefPtr<SharedBuffer> mBuffer;
|
||||
|
||||
size_t mBufferSize;
|
||||
|
||||
bool mFirstData;
|
||||
|
||||
RefPtr<PicoCallbackRunnable> mCallback;
|
||||
};
|
||||
|
||||
nsCOMPtr<nsIRunnable> sendEvent =
|
||||
new PicoSynthDataRunnable(aBuffer, aBufferSize, mFirstData, this);
|
||||
NS_DispatchToMainThread(sendEvent);
|
||||
mFirstData = false;
|
||||
}
|
||||
|
||||
// nsISpeechTaskCallback
|
||||
|
||||
NS_IMETHODIMP
|
||||
PicoCallbackRunnable::OnPause()
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
PicoCallbackRunnable::OnResume()
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
PicoCallbackRunnable::OnCancel()
|
||||
{
|
||||
mService->mCurrentTask = nullptr;
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
PicoCallbackRunnable::OnVolumeChanged(float aVolume)
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_INTERFACE_MAP_BEGIN(nsPicoService)
|
||||
NS_INTERFACE_MAP_ENTRY(nsISpeechService)
|
||||
NS_INTERFACE_MAP_ENTRY(nsIObserver)
|
||||
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsIObserver)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
NS_IMPL_ADDREF(nsPicoService)
|
||||
NS_IMPL_RELEASE(nsPicoService)
|
||||
|
||||
nsPicoService::nsPicoService()
|
||||
: mInitialized(false)
|
||||
, mVoicesMonitor("nsPicoService::mVoices")
|
||||
, mCurrentTask(nullptr)
|
||||
, mPicoSystem(nullptr)
|
||||
, mPicoEngine(nullptr)
|
||||
, mSgResource(nullptr)
|
||||
, mTaResource(nullptr)
|
||||
, mPicoMemArea(nullptr)
|
||||
{
|
||||
}
|
||||
|
||||
nsPicoService::~nsPicoService()
|
||||
{
|
||||
// We don't worry about removing the voices because this gets
|
||||
// destructed at shutdown along with the voice registry.
|
||||
MonitorAutoLock autoLock(mVoicesMonitor);
|
||||
mVoices.Clear();
|
||||
|
||||
if (mThread) {
|
||||
mThread->Shutdown();
|
||||
}
|
||||
|
||||
UnloadEngine();
|
||||
}
|
||||
|
||||
// nsIObserver
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsPicoService::Observe(nsISupports* aSubject, const char* aTopic,
|
||||
const char16_t* aData)
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
if(NS_WARN_IF(!(!strcmp(aTopic, "profile-after-change")))) {
|
||||
return NS_ERROR_UNEXPECTED;
|
||||
}
|
||||
|
||||
if (!Preferences::GetBool("media.webspeech.synth.enabled") ||
|
||||
Preferences::GetBool("media.webspeech.synth.test")) {
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
DebugOnly<nsresult> rv = NS_NewNamedThread("Pico Worker", getter_AddRefs(mThread));
|
||||
MOZ_ASSERT(NS_SUCCEEDED(rv));
|
||||
return mThread->Dispatch(
|
||||
NewRunnableMethod(this, &nsPicoService::Init), NS_DISPATCH_NORMAL);
|
||||
}
|
||||
// nsISpeechService
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsPicoService::Speak(const nsAString& aText, const nsAString& aUri,
|
||||
float aVolume, float aRate, float aPitch,
|
||||
nsISpeechTask* aTask)
|
||||
{
|
||||
if(NS_WARN_IF(!(mInitialized))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
MonitorAutoLock autoLock(mVoicesMonitor);
|
||||
bool found = false;
|
||||
PicoVoice* voice = mVoices.GetWeak(aUri, &found);
|
||||
if(NS_WARN_IF(!(found))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
mCurrentTask = aTask;
|
||||
RefPtr<PicoCallbackRunnable> cb = new PicoCallbackRunnable(aText, voice, aRate, aPitch, aTask, this);
|
||||
return mThread->Dispatch(cb, NS_DISPATCH_NORMAL);
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsPicoService::GetServiceType(SpeechServiceType* aServiceType)
|
||||
{
|
||||
*aServiceType = nsISpeechService::SERVICETYPE_DIRECT_AUDIO;
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
// private methods
|
||||
|
||||
void
|
||||
nsPicoService::Init()
|
||||
{
|
||||
MOZ_ASSERT(!NS_IsMainThread());
|
||||
MOZ_ASSERT(!mInitialized);
|
||||
|
||||
if (!sPicoApi.Init()) {
|
||||
NS_WARNING("Failed to initialize pico library");
|
||||
return;
|
||||
}
|
||||
|
||||
// Use environment variable, or default android/b2g path
|
||||
nsAutoCString langPath(PR_GetEnv("PICO_LANG_PATH"));
|
||||
|
||||
if (langPath.IsEmpty()) {
|
||||
langPath.AssignLiteral(GONK_PICO_LANG_PATH);
|
||||
}
|
||||
|
||||
nsCOMPtr<nsIFile> voicesDir;
|
||||
NS_NewNativeLocalFile(langPath, true, getter_AddRefs(voicesDir));
|
||||
|
||||
nsCOMPtr<nsISimpleEnumerator> dirIterator;
|
||||
nsresult rv = voicesDir->GetDirectoryEntries(getter_AddRefs(dirIterator));
|
||||
|
||||
if (NS_FAILED(rv)) {
|
||||
NS_WARNING(nsPrintfCString("Failed to get contents of directory: %s", langPath.get()).get());
|
||||
return;
|
||||
}
|
||||
|
||||
bool hasMoreElements = false;
|
||||
rv = dirIterator->HasMoreElements(&hasMoreElements);
|
||||
MOZ_ASSERT(NS_SUCCEEDED(rv));
|
||||
|
||||
MonitorAutoLock autoLock(mVoicesMonitor);
|
||||
|
||||
while (hasMoreElements && NS_SUCCEEDED(rv)) {
|
||||
nsCOMPtr<nsISupports> supports;
|
||||
rv = dirIterator->GetNext(getter_AddRefs(supports));
|
||||
MOZ_ASSERT(NS_SUCCEEDED(rv));
|
||||
|
||||
nsCOMPtr<nsIFile> voiceFile = do_QueryInterface(supports);
|
||||
MOZ_ASSERT(voiceFile);
|
||||
|
||||
nsAutoCString leafName;
|
||||
voiceFile->GetNativeLeafName(leafName);
|
||||
|
||||
nsAutoString lang;
|
||||
|
||||
if (GetVoiceFileLanguage(leafName, lang)) {
|
||||
nsAutoString uri;
|
||||
uri.AssignLiteral("urn:moz-tts:pico:");
|
||||
uri.Append(lang);
|
||||
|
||||
bool found = false;
|
||||
PicoVoice* voice = mVoices.GetWeak(uri, &found);
|
||||
|
||||
if (!found) {
|
||||
voice = new PicoVoice(lang);
|
||||
mVoices.Put(uri, voice);
|
||||
}
|
||||
|
||||
// Each voice consists of two lingware files: A language resource file,
|
||||
// suffixed by _ta.bin, and a speaker resource file, suffixed by _sb.bin.
|
||||
// We currently assume that there is a pair of files for each language.
|
||||
if (StringEndsWith(leafName, NS_LITERAL_CSTRING("_ta.bin"))) {
|
||||
rv = voiceFile->GetPersistentDescriptor(voice->mTaFile);
|
||||
MOZ_ASSERT(NS_SUCCEEDED(rv));
|
||||
} else if (StringEndsWith(leafName, NS_LITERAL_CSTRING("_sg.bin"))) {
|
||||
rv = voiceFile->GetPersistentDescriptor(voice->mSgFile);
|
||||
MOZ_ASSERT(NS_SUCCEEDED(rv));
|
||||
}
|
||||
}
|
||||
|
||||
rv = dirIterator->HasMoreElements(&hasMoreElements);
|
||||
}
|
||||
|
||||
NS_DispatchToMainThread(NewRunnableMethod(this, &nsPicoService::RegisterVoices));
|
||||
}
|
||||
|
||||
void
|
||||
nsPicoService::RegisterVoices()
|
||||
{
|
||||
nsSynthVoiceRegistry* registry = nsSynthVoiceRegistry::GetInstance();
|
||||
|
||||
for (auto iter = mVoices.Iter(); !iter.Done(); iter.Next()) {
|
||||
const nsAString& uri = iter.Key();
|
||||
RefPtr<PicoVoice>& voice = iter.Data();
|
||||
|
||||
// If we are missing either a language or a voice resource, it is invalid.
|
||||
if (voice->mTaFile.IsEmpty() || voice->mSgFile.IsEmpty()) {
|
||||
iter.Remove();
|
||||
continue;
|
||||
}
|
||||
|
||||
nsAutoString name;
|
||||
name.AssignLiteral("Pico ");
|
||||
name.Append(voice->mLanguage);
|
||||
|
||||
// This service is multi-threaded and can handle more than one utterance at a
|
||||
// time before previous utterances end. So, aQueuesUtterances == false
|
||||
DebugOnly<nsresult> rv =
|
||||
registry->AddVoice(this, uri, name, voice->mLanguage, true, false);
|
||||
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv), "Failed to add voice");
|
||||
}
|
||||
|
||||
mInitialized = true;
|
||||
}
|
||||
|
||||
bool
|
||||
nsPicoService::GetVoiceFileLanguage(const nsACString& aFileName, nsAString& aLang)
|
||||
{
|
||||
nsACString::const_iterator start, end;
|
||||
aFileName.BeginReading(start);
|
||||
aFileName.EndReading(end);
|
||||
|
||||
// The lingware filename syntax is language_(ta/sg).bin,
|
||||
// we extract the language prefix here.
|
||||
if (FindInReadable(NS_LITERAL_CSTRING("_"), start, end)) {
|
||||
end = start;
|
||||
aFileName.BeginReading(start);
|
||||
aLang.Assign(NS_ConvertUTF8toUTF16(Substring(start, end)));
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
void
|
||||
nsPicoService::LoadEngine(PicoVoice* aVoice)
|
||||
{
|
||||
PicoApi::pico_Status status = 0;
|
||||
|
||||
if (mPicoSystem) {
|
||||
UnloadEngine();
|
||||
}
|
||||
|
||||
if (!mPicoMemArea) {
|
||||
mPicoMemArea = MakeUnique<uint8_t[]>(PICO_MEM_SIZE);
|
||||
}
|
||||
|
||||
status = sPicoApi.pico_initialize(mPicoMemArea.get(),
|
||||
PICO_MEM_SIZE, &mPicoSystem);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_initialize", status);
|
||||
|
||||
status = sPicoApi.pico_loadResource(mPicoSystem, aVoice->mTaFile.get(), &mTaResource);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_loadResource", status);
|
||||
|
||||
status = sPicoApi.pico_loadResource(mPicoSystem, aVoice->mSgFile.get(), &mSgResource);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_loadResource", status);
|
||||
|
||||
status = sPicoApi.pico_createVoiceDefinition(mPicoSystem, PICO_VOICE_NAME);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_createVoiceDefinition", status);
|
||||
|
||||
char taName[PICO_RETSTRINGSIZE];
|
||||
status = sPicoApi.pico_getResourceName(mPicoSystem, mTaResource, taName);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_getResourceName", status);
|
||||
|
||||
status = sPicoApi.pico_addResourceToVoiceDefinition(
|
||||
mPicoSystem, PICO_VOICE_NAME, taName);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_addResourceToVoiceDefinition", status);
|
||||
|
||||
char sgName[PICO_RETSTRINGSIZE];
|
||||
status = sPicoApi.pico_getResourceName(mPicoSystem, mSgResource, sgName);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_getResourceName", status);
|
||||
|
||||
status = sPicoApi.pico_addResourceToVoiceDefinition(
|
||||
mPicoSystem, PICO_VOICE_NAME, sgName);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_addResourceToVoiceDefinition", status);
|
||||
|
||||
status = sPicoApi.pico_newEngine(mPicoSystem, PICO_VOICE_NAME, &mPicoEngine);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_newEngine", status);
|
||||
|
||||
if (sSingleton) {
|
||||
sSingleton->mCurrentVoice = aVoice;
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
nsPicoService::UnloadEngine()
|
||||
{
|
||||
PicoApi::pico_Status status = 0;
|
||||
|
||||
if (mPicoEngine) {
|
||||
status = sPicoApi.pico_disposeEngine(mPicoSystem, &mPicoEngine);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_disposeEngine", status);
|
||||
status = sPicoApi.pico_releaseVoiceDefinition(mPicoSystem, PICO_VOICE_NAME);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_releaseVoiceDefinition", status);
|
||||
mPicoEngine = nullptr;
|
||||
}
|
||||
|
||||
if (mSgResource) {
|
||||
status = sPicoApi.pico_unloadResource(mPicoSystem, &mSgResource);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_unloadResource", status);
|
||||
mSgResource = nullptr;
|
||||
}
|
||||
|
||||
if (mTaResource) {
|
||||
status = sPicoApi.pico_unloadResource(mPicoSystem, &mTaResource);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_unloadResource", status);
|
||||
mTaResource = nullptr;
|
||||
}
|
||||
|
||||
if (mPicoSystem) {
|
||||
status = sPicoApi.pico_terminate(&mPicoSystem);
|
||||
PICO_ENSURE_SUCCESS_VOID("pico_terminate", status);
|
||||
mPicoSystem = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
PicoVoice*
|
||||
nsPicoService::CurrentVoice()
|
||||
{
|
||||
MOZ_ASSERT(!NS_IsMainThread());
|
||||
|
||||
return mCurrentVoice;
|
||||
}
|
||||
|
||||
// static methods
|
||||
|
||||
nsPicoService*
|
||||
nsPicoService::GetInstance()
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
if (!XRE_IsParentProcess()) {
|
||||
MOZ_ASSERT(false, "nsPicoService can only be started on main gecko process");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
if (!sSingleton) {
|
||||
sSingleton = new nsPicoService();
|
||||
}
|
||||
|
||||
return sSingleton;
|
||||
}
|
||||
|
||||
already_AddRefed<nsPicoService>
|
||||
nsPicoService::GetInstanceForService()
|
||||
{
|
||||
RefPtr<nsPicoService> picoService = GetInstance();
|
||||
return picoService.forget();
|
||||
}
|
||||
|
||||
void
|
||||
nsPicoService::Shutdown()
|
||||
{
|
||||
if (!sSingleton) {
|
||||
return;
|
||||
}
|
||||
|
||||
sSingleton->mCurrentTask = nullptr;
|
||||
|
||||
sSingleton = nullptr;
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
93
dom/media/webspeech/synth/pico/nsPicoService.h
Normal file
93
dom/media/webspeech/synth/pico/nsPicoService.h
Normal file
|
|
@ -0,0 +1,93 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef nsPicoService_h
|
||||
#define nsPicoService_h
|
||||
|
||||
#include "mozilla/Mutex.h"
|
||||
#include "nsTArray.h"
|
||||
#include "nsIObserver.h"
|
||||
#include "nsIThread.h"
|
||||
#include "nsISpeechService.h"
|
||||
#include "nsRefPtrHashtable.h"
|
||||
#include "mozilla/StaticPtr.h"
|
||||
#include "mozilla/Monitor.h"
|
||||
#include "mozilla/UniquePtr.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class PicoVoice;
|
||||
class PicoCallbackRunnable;
|
||||
|
||||
typedef void* pico_System;
|
||||
typedef void* pico_Resource;
|
||||
typedef void* pico_Engine;
|
||||
|
||||
class nsPicoService : public nsIObserver,
|
||||
public nsISpeechService
|
||||
{
|
||||
friend class PicoCallbackRunnable;
|
||||
friend class PicoInitRunnable;
|
||||
|
||||
public:
|
||||
NS_DECL_THREADSAFE_ISUPPORTS
|
||||
NS_DECL_NSISPEECHSERVICE
|
||||
NS_DECL_NSIOBSERVER
|
||||
|
||||
nsPicoService();
|
||||
|
||||
static nsPicoService* GetInstance();
|
||||
|
||||
static already_AddRefed<nsPicoService> GetInstanceForService();
|
||||
|
||||
static void Shutdown();
|
||||
|
||||
private:
|
||||
|
||||
virtual ~nsPicoService();
|
||||
|
||||
void Init();
|
||||
|
||||
void RegisterVoices();
|
||||
|
||||
bool GetVoiceFileLanguage(const nsACString& aFileName, nsAString& aLang);
|
||||
|
||||
void LoadEngine(PicoVoice* aVoice);
|
||||
|
||||
void UnloadEngine();
|
||||
|
||||
PicoVoice* CurrentVoice();
|
||||
|
||||
bool mInitialized;
|
||||
|
||||
nsCOMPtr<nsIThread> mThread;
|
||||
|
||||
nsRefPtrHashtable<nsStringHashKey, PicoVoice> mVoices;
|
||||
|
||||
Monitor mVoicesMonitor;
|
||||
|
||||
PicoVoice* mCurrentVoice;
|
||||
|
||||
Atomic<nsISpeechTask*> mCurrentTask;
|
||||
|
||||
pico_System mPicoSystem;
|
||||
|
||||
pico_Engine mPicoEngine;
|
||||
|
||||
pico_Resource mSgResource;
|
||||
|
||||
pico_Resource mTaResource;
|
||||
|
||||
mozilla::UniquePtr<uint8_t[]> mPicoMemArea;
|
||||
|
||||
static StaticRefPtr<nsPicoService> sSingleton;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
56
dom/media/webspeech/synth/speechd/SpeechDispatcherModule.cpp
Normal file
56
dom/media/webspeech/synth/speechd/SpeechDispatcherModule.cpp
Normal file
|
|
@ -0,0 +1,56 @@
|
|||
/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim: set ts=8 sts=2 et sw=2 tw=80: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "mozilla/ModuleUtils.h"
|
||||
#include "nsIClassInfoImpl.h"
|
||||
#include "SpeechDispatcherService.h"
|
||||
|
||||
using namespace mozilla::dom;
|
||||
|
||||
#define SPEECHDISPATCHERSERVICE_CID \
|
||||
{0x8817b1cf, 0x5ada, 0x43bf, {0xbd, 0x73, 0x60, 0x76, 0x57, 0x70, 0x3d, 0x0d}}
|
||||
|
||||
#define SPEECHDISPATCHERSERVICE_CONTRACTID "@mozilla.org/synthspeechdispatcher;1"
|
||||
|
||||
// Defines SpeechDispatcherServiceConstructor
|
||||
NS_GENERIC_FACTORY_SINGLETON_CONSTRUCTOR(SpeechDispatcherService,
|
||||
SpeechDispatcherService::GetInstanceForService)
|
||||
|
||||
// Defines kSPEECHDISPATCHERSERVICE_CID
|
||||
NS_DEFINE_NAMED_CID(SPEECHDISPATCHERSERVICE_CID);
|
||||
|
||||
static const mozilla::Module::CIDEntry kCIDs[] = {
|
||||
{ &kSPEECHDISPATCHERSERVICE_CID, true, nullptr, SpeechDispatcherServiceConstructor },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static const mozilla::Module::ContractIDEntry kContracts[] = {
|
||||
{ SPEECHDISPATCHERSERVICE_CONTRACTID, &kSPEECHDISPATCHERSERVICE_CID },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static const mozilla::Module::CategoryEntry kCategories[] = {
|
||||
{ "speech-synth-started", "SpeechDispatcher Speech Synth", SPEECHDISPATCHERSERVICE_CONTRACTID },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static void
|
||||
UnloadSpeechDispatcherModule()
|
||||
{
|
||||
SpeechDispatcherService::Shutdown();
|
||||
}
|
||||
|
||||
static const mozilla::Module kModule = {
|
||||
mozilla::Module::kVersion,
|
||||
kCIDs,
|
||||
kContracts,
|
||||
kCategories,
|
||||
nullptr,
|
||||
nullptr,
|
||||
UnloadSpeechDispatcherModule
|
||||
};
|
||||
|
||||
NSMODULE_DEFN(synthspeechdispatcher) = &kModule;
|
||||
593
dom/media/webspeech/synth/speechd/SpeechDispatcherService.cpp
Normal file
593
dom/media/webspeech/synth/speechd/SpeechDispatcherService.cpp
Normal file
|
|
@ -0,0 +1,593 @@
|
|||
/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim: set ts=8 sts=2 et sw=2 tw=80: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "SpeechDispatcherService.h"
|
||||
|
||||
#include "mozilla/dom/nsSpeechTask.h"
|
||||
#include "mozilla/dom/nsSynthVoiceRegistry.h"
|
||||
#include "mozilla/Preferences.h"
|
||||
#include "nsEscape.h"
|
||||
#include "nsISupports.h"
|
||||
#include "nsPrintfCString.h"
|
||||
#include "nsReadableUtils.h"
|
||||
#include "nsServiceManagerUtils.h"
|
||||
#include "nsThreadUtils.h"
|
||||
#include "prlink.h"
|
||||
|
||||
#include <math.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
#define URI_PREFIX "urn:moz-tts:speechd:"
|
||||
|
||||
#define MAX_RATE static_cast<float>(2.5)
|
||||
#define MIN_RATE static_cast<float>(0.5)
|
||||
|
||||
// Some structures for libspeechd
|
||||
typedef enum {
|
||||
SPD_EVENT_BEGIN,
|
||||
SPD_EVENT_END,
|
||||
SPD_EVENT_INDEX_MARK,
|
||||
SPD_EVENT_CANCEL,
|
||||
SPD_EVENT_PAUSE,
|
||||
SPD_EVENT_RESUME
|
||||
} SPDNotificationType;
|
||||
|
||||
typedef enum {
|
||||
SPD_BEGIN = 1,
|
||||
SPD_END = 2,
|
||||
SPD_INDEX_MARKS = 4,
|
||||
SPD_CANCEL = 8,
|
||||
SPD_PAUSE = 16,
|
||||
SPD_RESUME = 32,
|
||||
|
||||
SPD_ALL = 0x3f
|
||||
} SPDNotification;
|
||||
|
||||
typedef enum {
|
||||
SPD_MODE_SINGLE = 0,
|
||||
SPD_MODE_THREADED = 1
|
||||
} SPDConnectionMode;
|
||||
|
||||
typedef void (*SPDCallback) (size_t msg_id, size_t client_id,
|
||||
SPDNotificationType state);
|
||||
|
||||
typedef void (*SPDCallbackIM) (size_t msg_id, size_t client_id,
|
||||
SPDNotificationType state, char* index_mark);
|
||||
|
||||
struct SPDConnection
|
||||
{
|
||||
SPDCallback callback_begin;
|
||||
SPDCallback callback_end;
|
||||
SPDCallback callback_cancel;
|
||||
SPDCallback callback_pause;
|
||||
SPDCallback callback_resume;
|
||||
SPDCallbackIM callback_im;
|
||||
|
||||
/* partial, more private fields in structure */
|
||||
};
|
||||
|
||||
struct SPDVoice
|
||||
{
|
||||
char* name;
|
||||
char* language;
|
||||
char* variant;
|
||||
};
|
||||
|
||||
typedef enum {
|
||||
SPD_IMPORTANT = 1,
|
||||
SPD_MESSAGE = 2,
|
||||
SPD_TEXT = 3,
|
||||
SPD_NOTIFICATION = 4,
|
||||
SPD_PROGRESS = 5
|
||||
} SPDPriority;
|
||||
|
||||
#define SPEECHD_FUNCTIONS \
|
||||
FUNC(spd_open, SPDConnection*, (const char*, const char*, const char*, SPDConnectionMode)) \
|
||||
FUNC(spd_close, void, (SPDConnection*)) \
|
||||
FUNC(spd_list_synthesis_voices, SPDVoice**, (SPDConnection*)) \
|
||||
FUNC(spd_say, int, (SPDConnection*, SPDPriority, const char*)) \
|
||||
FUNC(spd_cancel, int, (SPDConnection*)) \
|
||||
FUNC(spd_set_volume, int, (SPDConnection*, int)) \
|
||||
FUNC(spd_set_voice_rate, int, (SPDConnection*, int)) \
|
||||
FUNC(spd_set_voice_pitch, int, (SPDConnection*, int)) \
|
||||
FUNC(spd_set_synthesis_voice, int, (SPDConnection*, const char*)) \
|
||||
FUNC(spd_set_notification_on, int, (SPDConnection*, SPDNotification))
|
||||
|
||||
#define FUNC(name, type, params) \
|
||||
typedef type (*_##name##_fn) params; \
|
||||
static _##name##_fn _##name;
|
||||
|
||||
SPEECHD_FUNCTIONS
|
||||
|
||||
#undef FUNC
|
||||
|
||||
#define spd_open _spd_open
|
||||
#define spd_close _spd_close
|
||||
#define spd_list_synthesis_voices _spd_list_synthesis_voices
|
||||
#define spd_say _spd_say
|
||||
#define spd_cancel _spd_cancel
|
||||
#define spd_set_volume _spd_set_volume
|
||||
#define spd_set_voice_rate _spd_set_voice_rate
|
||||
#define spd_set_voice_pitch _spd_set_voice_pitch
|
||||
#define spd_set_synthesis_voice _spd_set_synthesis_voice
|
||||
#define spd_set_notification_on _spd_set_notification_on
|
||||
|
||||
static PRLibrary* speechdLib = nullptr;
|
||||
|
||||
typedef void (*nsSpeechDispatcherFunc)();
|
||||
struct nsSpeechDispatcherDynamicFunction
|
||||
{
|
||||
const char* functionName;
|
||||
nsSpeechDispatcherFunc* function;
|
||||
};
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
StaticRefPtr<SpeechDispatcherService> SpeechDispatcherService::sSingleton;
|
||||
|
||||
class SpeechDispatcherVoice
|
||||
{
|
||||
public:
|
||||
|
||||
SpeechDispatcherVoice(const nsAString& aName, const nsAString& aLanguage)
|
||||
: mName(aName), mLanguage(aLanguage) {}
|
||||
|
||||
NS_INLINE_DECL_THREADSAFE_REFCOUNTING(SpeechDispatcherVoice)
|
||||
|
||||
// Voice name
|
||||
nsString mName;
|
||||
|
||||
// Voice language, in BCP-47 syntax
|
||||
nsString mLanguage;
|
||||
|
||||
private:
|
||||
~SpeechDispatcherVoice() {}
|
||||
};
|
||||
|
||||
|
||||
class SpeechDispatcherCallback final : public nsISpeechTaskCallback
|
||||
{
|
||||
public:
|
||||
SpeechDispatcherCallback(nsISpeechTask* aTask, SpeechDispatcherService* aService)
|
||||
: mTask(aTask)
|
||||
, mService(aService) {}
|
||||
|
||||
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
|
||||
NS_DECL_CYCLE_COLLECTION_CLASS_AMBIGUOUS(SpeechDispatcherCallback, nsISpeechTaskCallback)
|
||||
|
||||
NS_DECL_NSISPEECHTASKCALLBACK
|
||||
|
||||
bool OnSpeechEvent(SPDNotificationType state);
|
||||
|
||||
private:
|
||||
~SpeechDispatcherCallback() { }
|
||||
|
||||
// This pointer is used to dispatch events
|
||||
nsCOMPtr<nsISpeechTask> mTask;
|
||||
|
||||
// By holding a strong reference to the service we guarantee that it won't be
|
||||
// destroyed before this runnable.
|
||||
RefPtr<SpeechDispatcherService> mService;
|
||||
|
||||
TimeStamp mStartTime;
|
||||
};
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION(SpeechDispatcherCallback, mTask);
|
||||
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechDispatcherCallback)
|
||||
NS_INTERFACE_MAP_ENTRY(nsISpeechTaskCallback)
|
||||
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsISpeechTaskCallback)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechDispatcherCallback)
|
||||
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechDispatcherCallback)
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechDispatcherCallback::OnPause()
|
||||
{
|
||||
// XXX: Speech dispatcher does not pause immediately, but waits for the speech
|
||||
// to reach an index mark so that it could resume from that offset.
|
||||
// There is no support for word or sentence boundaries, so index marks would
|
||||
// only occur in explicit SSML marks, and we don't support that yet.
|
||||
// What in actuality happens, is that if you call spd_pause(), it will speak
|
||||
// the utterance in its entirety, dispatch an end event, and then put speechd
|
||||
// in a 'paused' state. Since it is after the utterance ended, we don't get
|
||||
// that state change, and our speech api is in an unrecoverable state.
|
||||
// So, since it is useless anyway, I am not implementing pause.
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechDispatcherCallback::OnResume()
|
||||
{
|
||||
// XXX: Unsupported, see OnPause().
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechDispatcherCallback::OnCancel()
|
||||
{
|
||||
if (spd_cancel(mService->mSpeechdClient) < 0) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechDispatcherCallback::OnVolumeChanged(float aVolume)
|
||||
{
|
||||
// XXX: This currently does not change the volume mid-utterance, but it
|
||||
// doesn't do anything bad either. So we could put this here with the hopes
|
||||
// that speechd supports this in the future.
|
||||
if (spd_set_volume(mService->mSpeechdClient, static_cast<int>(aVolume * 100)) < 0) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
bool
|
||||
SpeechDispatcherCallback::OnSpeechEvent(SPDNotificationType state)
|
||||
{
|
||||
bool remove = false;
|
||||
|
||||
switch (state) {
|
||||
case SPD_EVENT_BEGIN:
|
||||
mStartTime = TimeStamp::Now();
|
||||
mTask->DispatchStart();
|
||||
break;
|
||||
|
||||
case SPD_EVENT_PAUSE:
|
||||
mTask->DispatchPause((TimeStamp::Now() - mStartTime).ToSeconds(), 0);
|
||||
break;
|
||||
|
||||
case SPD_EVENT_RESUME:
|
||||
mTask->DispatchResume((TimeStamp::Now() - mStartTime).ToSeconds(), 0);
|
||||
break;
|
||||
|
||||
case SPD_EVENT_CANCEL:
|
||||
case SPD_EVENT_END:
|
||||
mTask->DispatchEnd((TimeStamp::Now() - mStartTime).ToSeconds(), 0);
|
||||
remove = true;
|
||||
break;
|
||||
|
||||
case SPD_EVENT_INDEX_MARK:
|
||||
// Not yet supported
|
||||
break;
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return remove;
|
||||
}
|
||||
|
||||
static void
|
||||
speechd_cb(size_t msg_id, size_t client_id, SPDNotificationType state)
|
||||
{
|
||||
SpeechDispatcherService* service = SpeechDispatcherService::GetInstance(false);
|
||||
|
||||
if (service) {
|
||||
NS_DispatchToMainThread(
|
||||
NewRunnableMethod<uint32_t, SPDNotificationType>(
|
||||
service, &SpeechDispatcherService::EventNotify,
|
||||
static_cast<uint32_t>(msg_id), state));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
NS_INTERFACE_MAP_BEGIN(SpeechDispatcherService)
|
||||
NS_INTERFACE_MAP_ENTRY(nsISpeechService)
|
||||
NS_INTERFACE_MAP_ENTRY(nsIObserver)
|
||||
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsIObserver)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
NS_IMPL_ADDREF(SpeechDispatcherService)
|
||||
NS_IMPL_RELEASE(SpeechDispatcherService)
|
||||
|
||||
SpeechDispatcherService::SpeechDispatcherService()
|
||||
: mInitialized(false)
|
||||
, mSpeechdClient(nullptr)
|
||||
{
|
||||
}
|
||||
|
||||
void
|
||||
SpeechDispatcherService::Init()
|
||||
{
|
||||
if (!Preferences::GetBool("media.webspeech.synth.enabled") ||
|
||||
Preferences::GetBool("media.webspeech.synth.test")) {
|
||||
return;
|
||||
}
|
||||
|
||||
// While speech dispatcher has a "threaded" mode, only spd_say() is async.
|
||||
// Since synchronous socket i/o could impact startup time, we do
|
||||
// initialization in a separate thread.
|
||||
DebugOnly<nsresult> rv = NS_NewNamedThread("speechd init",
|
||||
getter_AddRefs(mInitThread));
|
||||
MOZ_ASSERT(NS_SUCCEEDED(rv));
|
||||
rv = mInitThread->Dispatch(
|
||||
NewRunnableMethod(this, &SpeechDispatcherService::Setup), NS_DISPATCH_NORMAL);
|
||||
MOZ_ASSERT(NS_SUCCEEDED(rv));
|
||||
}
|
||||
|
||||
SpeechDispatcherService::~SpeechDispatcherService()
|
||||
{
|
||||
if (mInitThread) {
|
||||
mInitThread->Shutdown();
|
||||
}
|
||||
|
||||
if (mSpeechdClient) {
|
||||
spd_close(mSpeechdClient);
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
SpeechDispatcherService::Setup()
|
||||
{
|
||||
#define FUNC(name, type, params) { #name, (nsSpeechDispatcherFunc *)&_##name },
|
||||
static const nsSpeechDispatcherDynamicFunction kSpeechDispatcherSymbols[] = {
|
||||
SPEECHD_FUNCTIONS
|
||||
};
|
||||
#undef FUNC
|
||||
|
||||
MOZ_ASSERT(!mInitialized);
|
||||
|
||||
speechdLib = PR_LoadLibrary("libspeechd.so.2");
|
||||
|
||||
if (!speechdLib) {
|
||||
NS_WARNING("Failed to load speechd library");
|
||||
return;
|
||||
}
|
||||
|
||||
if (!PR_FindFunctionSymbol(speechdLib, "spd_get_volume")) {
|
||||
// There is no version getter function, so we rely on a symbol that was
|
||||
// introduced in release 0.8.2 in order to check for ABI compatibility.
|
||||
NS_WARNING("Unsupported version of speechd detected");
|
||||
return;
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < ArrayLength(kSpeechDispatcherSymbols); i++) {
|
||||
*kSpeechDispatcherSymbols[i].function =
|
||||
PR_FindFunctionSymbol(speechdLib, kSpeechDispatcherSymbols[i].functionName);
|
||||
|
||||
if (!*kSpeechDispatcherSymbols[i].function) {
|
||||
NS_WARNING(nsPrintfCString("Failed to find speechd symbol for'%s'",
|
||||
kSpeechDispatcherSymbols[i].functionName).get());
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
mSpeechdClient = spd_open("firefox", "web speech api", "who", SPD_MODE_THREADED);
|
||||
if (!mSpeechdClient) {
|
||||
NS_WARNING("Failed to call spd_open");
|
||||
return;
|
||||
}
|
||||
|
||||
// Get all the voices from sapi and register in the SynthVoiceRegistry
|
||||
SPDVoice** list = spd_list_synthesis_voices(mSpeechdClient);
|
||||
|
||||
mSpeechdClient->callback_begin = speechd_cb;
|
||||
mSpeechdClient->callback_end = speechd_cb;
|
||||
mSpeechdClient->callback_cancel = speechd_cb;
|
||||
mSpeechdClient->callback_pause = speechd_cb;
|
||||
mSpeechdClient->callback_resume = speechd_cb;
|
||||
|
||||
spd_set_notification_on(mSpeechdClient, SPD_BEGIN);
|
||||
spd_set_notification_on(mSpeechdClient, SPD_END);
|
||||
spd_set_notification_on(mSpeechdClient, SPD_CANCEL);
|
||||
|
||||
if (list != NULL) {
|
||||
for (int i = 0; list[i]; i++) {
|
||||
nsAutoString uri;
|
||||
|
||||
uri.AssignLiteral(URI_PREFIX);
|
||||
nsAutoCString name;
|
||||
NS_EscapeURL(list[i]->name, -1, esc_OnlyNonASCII | esc_AlwaysCopy, name);
|
||||
uri.Append(NS_ConvertUTF8toUTF16(name));;
|
||||
uri.AppendLiteral("?");
|
||||
|
||||
nsAutoCString lang(list[i]->language);
|
||||
|
||||
if (strcmp(list[i]->variant, "none") != 0) {
|
||||
// In speech dispatcher, the variant will usually be the locale subtag
|
||||
// with another, non-standard suptag after it. We keep the first one
|
||||
// and convert it to uppercase.
|
||||
const char* v = list[i]->variant;
|
||||
const char* hyphen = strchr(v, '-');
|
||||
nsDependentCSubstring variant(v, hyphen ? hyphen - v : strlen(v));
|
||||
ToUpperCase(variant);
|
||||
|
||||
// eSpeak uses UK which is not a valid region subtag in BCP47.
|
||||
if (variant.Equals("UK")) {
|
||||
variant.AssignLiteral("GB");
|
||||
}
|
||||
|
||||
lang.AppendLiteral("-");
|
||||
lang.Append(variant);
|
||||
}
|
||||
|
||||
uri.Append(NS_ConvertUTF8toUTF16(lang));
|
||||
|
||||
mVoices.Put(uri, new SpeechDispatcherVoice(
|
||||
NS_ConvertUTF8toUTF16(list[i]->name),
|
||||
NS_ConvertUTF8toUTF16(lang)));
|
||||
}
|
||||
}
|
||||
|
||||
NS_DispatchToMainThread(NewRunnableMethod(this, &SpeechDispatcherService::RegisterVoices));
|
||||
|
||||
//mInitialized = true;
|
||||
}
|
||||
|
||||
// private methods
|
||||
|
||||
void
|
||||
SpeechDispatcherService::RegisterVoices()
|
||||
{
|
||||
RefPtr<nsSynthVoiceRegistry> registry = nsSynthVoiceRegistry::GetInstance();
|
||||
for (auto iter = mVoices.Iter(); !iter.Done(); iter.Next()) {
|
||||
RefPtr<SpeechDispatcherVoice>& voice = iter.Data();
|
||||
|
||||
// This service can only speak one utterance at a time, so we set
|
||||
// aQueuesUtterances to true in order to track global state and schedule
|
||||
// access to this service.
|
||||
DebugOnly<nsresult> rv =
|
||||
registry->AddVoice(this, iter.Key(), voice->mName, voice->mLanguage,
|
||||
voice->mName.EqualsLiteral("default"), true);
|
||||
|
||||
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv), "Failed to add voice");
|
||||
}
|
||||
|
||||
mInitThread->Shutdown();
|
||||
mInitThread = nullptr;
|
||||
|
||||
mInitialized = true;
|
||||
|
||||
registry->NotifyVoicesChanged();
|
||||
}
|
||||
|
||||
// nsIObserver
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechDispatcherService::Observe(nsISupports* aSubject, const char* aTopic,
|
||||
const char16_t* aData)
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
// nsISpeechService
|
||||
|
||||
// TODO: Support SSML
|
||||
NS_IMETHODIMP
|
||||
SpeechDispatcherService::Speak(const nsAString& aText, const nsAString& aUri,
|
||||
float aVolume, float aRate, float aPitch,
|
||||
nsISpeechTask* aTask)
|
||||
{
|
||||
if (NS_WARN_IF(!mInitialized)) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
RefPtr<SpeechDispatcherCallback> callback =
|
||||
new SpeechDispatcherCallback(aTask, this);
|
||||
|
||||
bool found = false;
|
||||
SpeechDispatcherVoice* voice = mVoices.GetWeak(aUri, &found);
|
||||
|
||||
if(NS_WARN_IF(!(found))) {
|
||||
return NS_ERROR_NOT_AVAILABLE;
|
||||
}
|
||||
|
||||
spd_set_synthesis_voice(mSpeechdClient,
|
||||
NS_ConvertUTF16toUTF8(voice->mName).get());
|
||||
|
||||
// We provide a volume of 0.0 to 1.0, speech-dispatcher expects 0 - 100.
|
||||
spd_set_volume(mSpeechdClient, static_cast<int>(aVolume * 100));
|
||||
|
||||
// aRate is a value of 0.1 (0.1x) to 10 (10x) with 1 (1x) being normal rate.
|
||||
// speechd expects -100 to 100 with 0 being normal rate.
|
||||
float rate = 0;
|
||||
if (aRate > 1) {
|
||||
// Each step to 100 is logarithmically distributed up to 2.5x.
|
||||
rate = log10(std::min(aRate, MAX_RATE)) / log10(MAX_RATE) * 100;
|
||||
} else if (aRate < 1) {
|
||||
// Each step to -100 is logarithmically distributed down to 0.5x.
|
||||
rate = log10(std::max(aRate, MIN_RATE)) / log10(MIN_RATE) * -100;
|
||||
}
|
||||
|
||||
spd_set_voice_rate(mSpeechdClient, static_cast<int>(rate));
|
||||
|
||||
// We provide a pitch of 0 to 2 with 1 being the default.
|
||||
// speech-dispatcher expects -100 to 100 with 0 being default.
|
||||
spd_set_voice_pitch(mSpeechdClient, static_cast<int>((aPitch - 1) * 100));
|
||||
|
||||
// The last three parameters don't matter for an indirect service
|
||||
nsresult rv = aTask->Setup(callback, 0, 0, 0);
|
||||
|
||||
if (NS_FAILED(rv)) {
|
||||
return rv;
|
||||
}
|
||||
|
||||
if (aText.Length()) {
|
||||
int msg_id = spd_say(
|
||||
mSpeechdClient, SPD_MESSAGE, NS_ConvertUTF16toUTF8(aText).get());
|
||||
|
||||
if (msg_id < 0) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
mCallbacks.Put(msg_id, callback);
|
||||
} else {
|
||||
// Speech dispatcher does not work well with empty strings.
|
||||
// In that case, don't send empty string to speechd,
|
||||
// and just emulate a speechd start and end event.
|
||||
NS_DispatchToMainThread(NewRunnableMethod<SPDNotificationType>(
|
||||
callback, &SpeechDispatcherCallback::OnSpeechEvent, SPD_EVENT_BEGIN));
|
||||
|
||||
NS_DispatchToMainThread(NewRunnableMethod<SPDNotificationType>(
|
||||
callback, &SpeechDispatcherCallback::OnSpeechEvent, SPD_EVENT_END));
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
SpeechDispatcherService::GetServiceType(SpeechServiceType* aServiceType)
|
||||
{
|
||||
*aServiceType = nsISpeechService::SERVICETYPE_INDIRECT_AUDIO;
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
SpeechDispatcherService*
|
||||
SpeechDispatcherService::GetInstance(bool create)
|
||||
{
|
||||
if (XRE_GetProcessType() != GeckoProcessType_Default) {
|
||||
MOZ_ASSERT(false,
|
||||
"SpeechDispatcherService can only be started on main gecko process");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
if (!sSingleton && create) {
|
||||
sSingleton = new SpeechDispatcherService();
|
||||
sSingleton->Init();
|
||||
}
|
||||
|
||||
return sSingleton;
|
||||
}
|
||||
|
||||
already_AddRefed<SpeechDispatcherService>
|
||||
SpeechDispatcherService::GetInstanceForService()
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
RefPtr<SpeechDispatcherService> sapiService = GetInstance();
|
||||
return sapiService.forget();
|
||||
}
|
||||
|
||||
void
|
||||
SpeechDispatcherService::EventNotify(uint32_t aMsgId, uint32_t aState)
|
||||
{
|
||||
SpeechDispatcherCallback* callback = mCallbacks.GetWeak(aMsgId);
|
||||
|
||||
if (callback) {
|
||||
if (callback->OnSpeechEvent((SPDNotificationType)aState)) {
|
||||
mCallbacks.Remove(aMsgId);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
SpeechDispatcherService::Shutdown()
|
||||
{
|
||||
if (!sSingleton) {
|
||||
return;
|
||||
}
|
||||
|
||||
sSingleton = nullptr;
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
67
dom/media/webspeech/synth/speechd/SpeechDispatcherService.h
Normal file
67
dom/media/webspeech/synth/speechd/SpeechDispatcherService.h
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim: set ts=8 sts=2 et sw=2 tw=80: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef mozilla_dom_SpeechDispatcherService_h
|
||||
#define mozilla_dom_SpeechDispatcherService_h
|
||||
|
||||
#include "mozilla/StaticPtr.h"
|
||||
#include "nsIObserver.h"
|
||||
#include "nsISpeechService.h"
|
||||
#include "nsIThread.h"
|
||||
#include "nsRefPtrHashtable.h"
|
||||
#include "nsTArray.h"
|
||||
|
||||
struct SPDConnection;
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class SpeechDispatcherCallback;
|
||||
class SpeechDispatcherVoice;
|
||||
|
||||
class SpeechDispatcherService final : public nsIObserver,
|
||||
public nsISpeechService
|
||||
{
|
||||
friend class SpeechDispatcherCallback;
|
||||
public:
|
||||
NS_DECL_THREADSAFE_ISUPPORTS
|
||||
NS_DECL_NSIOBSERVER
|
||||
NS_DECL_NSISPEECHSERVICE
|
||||
|
||||
SpeechDispatcherService();
|
||||
|
||||
void Init();
|
||||
|
||||
void Setup();
|
||||
|
||||
void EventNotify(uint32_t aMsgId, uint32_t aState);
|
||||
|
||||
static SpeechDispatcherService* GetInstance(bool create = true);
|
||||
static already_AddRefed<SpeechDispatcherService> GetInstanceForService();
|
||||
|
||||
static void Shutdown();
|
||||
|
||||
static StaticRefPtr<SpeechDispatcherService> sSingleton;
|
||||
|
||||
private:
|
||||
virtual ~SpeechDispatcherService();
|
||||
|
||||
void RegisterVoices();
|
||||
|
||||
bool mInitialized;
|
||||
|
||||
SPDConnection* mSpeechdClient;
|
||||
|
||||
nsRefPtrHashtable<nsUint32HashKey, SpeechDispatcherCallback> mCallbacks;
|
||||
|
||||
nsCOMPtr<nsIThread> mInitThread;
|
||||
|
||||
nsRefPtrHashtable<nsStringHashKey, SpeechDispatcherVoice> mVoices;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
#endif
|
||||
13
dom/media/webspeech/synth/speechd/moz.build
Normal file
13
dom/media/webspeech/synth/speechd/moz.build
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
# -*- Mode: python; indent-tabs-mode: nil; tab-width: 40 -*-
|
||||
# vim: set filetype=python:
|
||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
UNIFIED_SOURCES += [
|
||||
'SpeechDispatcherModule.cpp',
|
||||
'SpeechDispatcherService.cpp'
|
||||
]
|
||||
include('/ipc/chromium/chromium-config.mozbuild')
|
||||
|
||||
FINAL_LIBRARY = 'xul'
|
||||
55
dom/media/webspeech/synth/test/FakeSynthModule.cpp
Normal file
55
dom/media/webspeech/synth/test/FakeSynthModule.cpp
Normal file
|
|
@ -0,0 +1,55 @@
|
|||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "mozilla/ModuleUtils.h"
|
||||
#include "nsIClassInfoImpl.h"
|
||||
|
||||
#include "nsFakeSynthServices.h"
|
||||
|
||||
using namespace mozilla::dom;
|
||||
|
||||
#define FAKESYNTHSERVICE_CID \
|
||||
{0xe7d52d9e, 0xc148, 0x47d8, {0xab, 0x2a, 0x95, 0xd7, 0xf4, 0x0e, 0xa5, 0x3d}}
|
||||
|
||||
#define FAKESYNTHSERVICE_CONTRACTID "@mozilla.org/fakesynth;1"
|
||||
|
||||
// Defines nsFakeSynthServicesConstructor
|
||||
NS_GENERIC_FACTORY_SINGLETON_CONSTRUCTOR(nsFakeSynthServices,
|
||||
nsFakeSynthServices::GetInstanceForService)
|
||||
|
||||
// Defines kFAKESYNTHSERVICE_CID
|
||||
NS_DEFINE_NAMED_CID(FAKESYNTHSERVICE_CID);
|
||||
|
||||
static const mozilla::Module::CIDEntry kCIDs[] = {
|
||||
{ &kFAKESYNTHSERVICE_CID, true, nullptr, nsFakeSynthServicesConstructor },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static const mozilla::Module::ContractIDEntry kContracts[] = {
|
||||
{ FAKESYNTHSERVICE_CONTRACTID, &kFAKESYNTHSERVICE_CID },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static const mozilla::Module::CategoryEntry kCategories[] = {
|
||||
{ "speech-synth-started", "Fake Speech Synth", FAKESYNTHSERVICE_CONTRACTID },
|
||||
{ nullptr }
|
||||
};
|
||||
|
||||
static void
|
||||
UnloadFakeSynthmodule()
|
||||
{
|
||||
nsFakeSynthServices::Shutdown();
|
||||
}
|
||||
|
||||
static const mozilla::Module kModule = {
|
||||
mozilla::Module::kVersion,
|
||||
kCIDs,
|
||||
kContracts,
|
||||
kCategories,
|
||||
nullptr,
|
||||
nullptr,
|
||||
UnloadFakeSynthmodule
|
||||
};
|
||||
|
||||
NSMODULE_DEFN(fakesynth) = &kModule;
|
||||
91
dom/media/webspeech/synth/test/common.js
Normal file
91
dom/media/webspeech/synth/test/common.js
Normal file
|
|
@ -0,0 +1,91 @@
|
|||
function synthTestQueue(aTestArgs, aEndFunc) {
|
||||
var utterances = [];
|
||||
for (var i in aTestArgs) {
|
||||
var uargs = aTestArgs[i][0];
|
||||
var win = uargs.win || window;
|
||||
var u = new win.SpeechSynthesisUtterance(uargs.text);
|
||||
|
||||
if (uargs.args) {
|
||||
for (var attr in uargs.args)
|
||||
u[attr] = uargs.args[attr];
|
||||
}
|
||||
|
||||
function onend_handler(e) {
|
||||
is(e.target, utterances.shift(), "Target matches utterances");
|
||||
ok(!speechSynthesis.speaking, "speechSynthesis is not speaking.");
|
||||
|
||||
if (utterances.length) {
|
||||
ok(speechSynthesis.pending, "other utterances queued");
|
||||
} else {
|
||||
ok(!speechSynthesis.pending, "queue is empty, nothing pending.");
|
||||
if (aEndFunc)
|
||||
aEndFunc();
|
||||
}
|
||||
}
|
||||
|
||||
u.addEventListener('start',
|
||||
(function (expectedUri) {
|
||||
return function (e) {
|
||||
if (expectedUri) {
|
||||
var chosenVoice = SpecialPowers.wrap(e).target.chosenVoiceURI;
|
||||
is(chosenVoice, expectedUri, "Incorrect URI is used");
|
||||
}
|
||||
};
|
||||
})(aTestArgs[i][1] ? aTestArgs[i][1].uri : null));
|
||||
|
||||
u.addEventListener('end', onend_handler);
|
||||
u.addEventListener('error', onend_handler);
|
||||
|
||||
u.addEventListener('error',
|
||||
(function (expectedError) {
|
||||
return function onerror_handler(e) {
|
||||
ok(expectedError, "Error in speech utterance '" + e.target.text + "'");
|
||||
};
|
||||
})(aTestArgs[i][1] ? aTestArgs[i][1].err : false));
|
||||
|
||||
utterances.push(u);
|
||||
win.speechSynthesis.speak(u);
|
||||
}
|
||||
|
||||
ok(!speechSynthesis.speaking, "speechSynthesis is not speaking yet.");
|
||||
ok(speechSynthesis.pending, "speechSynthesis has an utterance queued.");
|
||||
}
|
||||
|
||||
function loadFrame(frameId) {
|
||||
return new Promise(function(resolve, reject) {
|
||||
var frame = document.getElementById(frameId);
|
||||
frame.addEventListener('load', function (e) {
|
||||
frame.contentWindow.document.title = frameId;
|
||||
resolve(frame);
|
||||
});
|
||||
frame.src = 'data:text/html,' + encodeURI('<html><head></head><body></body></html>');
|
||||
});
|
||||
}
|
||||
|
||||
function waitForVoices(win) {
|
||||
return new Promise(resolve => {
|
||||
function resolver() {
|
||||
if (win.speechSynthesis.getVoices().length) {
|
||||
win.speechSynthesis.removeEventListener('voiceschanged', resolver);
|
||||
resolve();
|
||||
}
|
||||
}
|
||||
|
||||
win.speechSynthesis.addEventListener('voiceschanged', resolver);
|
||||
resolver();
|
||||
});
|
||||
}
|
||||
|
||||
function loadSpeechTest(fileName, prefs, frameId="testFrame") {
|
||||
loadFrame(frameId).then(frame => {
|
||||
waitForVoices(frame.contentWindow).then(
|
||||
() => document.getElementById("testFrame").src = fileName);
|
||||
});
|
||||
}
|
||||
|
||||
function testSynthState(win, expectedState) {
|
||||
for (var attr in expectedState) {
|
||||
is(win.speechSynthesis[attr], expectedState[attr],
|
||||
win.document.title + ": '" + attr + '" does not match');
|
||||
}
|
||||
}
|
||||
28
dom/media/webspeech/synth/test/file_bfcache_frame.html
Normal file
28
dom/media/webspeech/synth/test/file_bfcache_frame.html
Normal file
|
|
@ -0,0 +1,28 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<script type="application/javascript">
|
||||
var frameUnloaded = function() {
|
||||
var u = new SpeechSynthesisUtterance('hi');
|
||||
u.addEventListener('end', function () {
|
||||
parent.ok(true, 'Successfully spoke utterance from new frame.');
|
||||
parent.onDone();
|
||||
});
|
||||
speechSynthesis.speak(u);
|
||||
}
|
||||
addEventListener('pageshow', function onshow(evt) {
|
||||
var u = new SpeechSynthesisUtterance('hello');
|
||||
u.lang = 'it-IT-noend';
|
||||
u.addEventListener('start', function() {
|
||||
location =
|
||||
'data:text/html,<html><body onpageshow="' +
|
||||
frameUnloaded.toSource() + '()"></body></html>';
|
||||
});
|
||||
speechSynthesis.speak(u);
|
||||
});
|
||||
</script>
|
||||
</head>
|
||||
<body>
|
||||
</body>
|
||||
</html>
|
||||
69
dom/media/webspeech/synth/test/file_global_queue.html
Normal file
69
dom/media/webspeech/synth/test/file_global_queue.html
Normal file
|
|
@ -0,0 +1,69 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=1188099
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 1188099: Global queue should correctly schedule utterances</title>
|
||||
<script type="application/javascript">
|
||||
window.SimpleTest = parent.SimpleTest;
|
||||
window.info = parent.info;
|
||||
window.is = parent.is;
|
||||
window.isnot = parent.isnot;
|
||||
window.ok = parent.ok;
|
||||
window.todo = parent.todo;
|
||||
</script>
|
||||
<script type="application/javascript" src="common.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1188099">Mozilla Bug 1188099</a>
|
||||
<iframe id="frame1"></iframe>
|
||||
<iframe id="frame2"></iframe>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="application/javascript">
|
||||
Promise.all([loadFrame('frame1'), loadFrame('frame2')]).then(function ([frame1, frame2]) {
|
||||
var win1 = frame1.contentWindow;
|
||||
var win2 = frame2.contentWindow;
|
||||
var utterance1 = new win1.SpeechSynthesisUtterance("hello, losers");
|
||||
var utterance2 = new win1.SpeechSynthesisUtterance("hello, losers three");
|
||||
var utterance3 = new win2.SpeechSynthesisUtterance("hello, losers too");
|
||||
var eventOrder = ['start1', 'end1', 'start3', 'end3', 'start2', 'end2'];
|
||||
utterance1.addEventListener('start', function(e) {
|
||||
is(eventOrder.shift(), 'start1', 'start1');
|
||||
testSynthState(win1, { speaking: true, pending: true });
|
||||
testSynthState(win2, { speaking: true, pending: true });
|
||||
});
|
||||
utterance1.addEventListener('end', function(e) {
|
||||
is(eventOrder.shift(), 'end1', 'end1');
|
||||
});
|
||||
utterance3.addEventListener('start', function(e) {
|
||||
is(eventOrder.shift(), 'start3', 'start3');
|
||||
testSynthState(win1, { speaking: true, pending: true });
|
||||
testSynthState(win2, { speaking: true, pending: false });
|
||||
});
|
||||
utterance3.addEventListener('end', function(e) {
|
||||
is(eventOrder.shift(), 'end3', 'end3');
|
||||
});
|
||||
utterance2.addEventListener('start', function(e) {
|
||||
is(eventOrder.shift(), 'start2', 'start2');
|
||||
testSynthState(win1, { speaking: true, pending: false });
|
||||
testSynthState(win2, { speaking: true, pending: false });
|
||||
});
|
||||
utterance2.addEventListener('end', function(e) {
|
||||
is(eventOrder.shift(), 'end2', 'end2');
|
||||
testSynthState(win1, { speaking: false, pending: false });
|
||||
testSynthState(win2, { speaking: false, pending: false });
|
||||
SimpleTest.finish();
|
||||
});
|
||||
win1.speechSynthesis.speak(utterance1);
|
||||
win1.speechSynthesis.speak(utterance2);
|
||||
win2.speechSynthesis.speak(utterance3);
|
||||
});
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
88
dom/media/webspeech/synth/test/file_global_queue_cancel.html
Normal file
88
dom/media/webspeech/synth/test/file_global_queue_cancel.html
Normal file
|
|
@ -0,0 +1,88 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=1188099
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 1188099: Calling cancel() should work correctly with global queue</title>
|
||||
<script type="application/javascript">
|
||||
window.SimpleTest = parent.SimpleTest;
|
||||
window.info = parent.info;
|
||||
window.is = parent.is;
|
||||
window.isnot = parent.isnot;
|
||||
window.ok = parent.ok;
|
||||
window.todo = parent.todo;
|
||||
</script>
|
||||
<script type="application/javascript" src="common.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1188099">Mozilla Bug 1188099</a>
|
||||
<iframe id="frame1"></iframe>
|
||||
<iframe id="frame2"></iframe>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="application/javascript">
|
||||
Promise.all([loadFrame('frame1'), loadFrame('frame2')]).then(function ([frame1, frame2]) {
|
||||
var win1 = frame1.contentWindow;
|
||||
var win2 = frame2.contentWindow;
|
||||
|
||||
var utterance1 = new win1.SpeechSynthesisUtterance(
|
||||
"u1: Donec ac nunc feugiat, posuere");
|
||||
utterance1.lang = 'it-IT-noend';
|
||||
var utterance2 = new win1.SpeechSynthesisUtterance("u2: hello, losers too");
|
||||
utterance2.lang = 'it-IT-noend';
|
||||
var utterance3 = new win1.SpeechSynthesisUtterance("u3: hello, losers three");
|
||||
|
||||
var utterance4 = new win2.SpeechSynthesisUtterance("u4: hello, losers same!");
|
||||
utterance4.lang = 'it-IT-noend';
|
||||
var utterance5 = new win2.SpeechSynthesisUtterance("u5: hello, losers too");
|
||||
utterance5.lang = 'it-IT-noend';
|
||||
|
||||
var eventOrder = ['start1', 'end1', 'start2', 'end2'];
|
||||
utterance1.addEventListener('start', function(e) {
|
||||
is(eventOrder.shift(), 'start1', 'start1');
|
||||
testSynthState(win1, { speaking: true, pending: true });
|
||||
testSynthState(win2, { speaking: true, pending: true });
|
||||
win2.speechSynthesis.cancel();
|
||||
SpecialPowers.wrap(win1.speechSynthesis).forceEnd();
|
||||
|
||||
});
|
||||
utterance1.addEventListener('end', function(e) {
|
||||
is(eventOrder.shift(), 'end1', 'end1');
|
||||
testSynthState(win1, { pending: true });
|
||||
testSynthState(win2, { pending: false });
|
||||
});
|
||||
utterance2.addEventListener('start', function(e) {
|
||||
is(eventOrder.shift(), 'start2', 'start2');
|
||||
testSynthState(win1, { speaking: true, pending: true });
|
||||
testSynthState(win2, { speaking: true, pending: false });
|
||||
win1.speechSynthesis.cancel();
|
||||
});
|
||||
utterance2.addEventListener('end', function(e) {
|
||||
is(eventOrder.shift(), 'end2', 'end2');
|
||||
testSynthState(win1, { speaking: false, pending: false });
|
||||
testSynthState(win2, { speaking: false, pending: false });
|
||||
SimpleTest.finish();
|
||||
});
|
||||
|
||||
function wrongUtterance(e) {
|
||||
ok(false, 'This shall not be uttered: "' + e.target.text + '"');
|
||||
}
|
||||
|
||||
utterance3.addEventListener('start', wrongUtterance);
|
||||
utterance4.addEventListener('start', wrongUtterance);
|
||||
utterance5.addEventListener('start', wrongUtterance);
|
||||
|
||||
win1.speechSynthesis.speak(utterance1);
|
||||
win1.speechSynthesis.speak(utterance2);
|
||||
win1.speechSynthesis.speak(utterance3);
|
||||
win2.speechSynthesis.speak(utterance4);
|
||||
win2.speechSynthesis.speak(utterance5);
|
||||
});
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
131
dom/media/webspeech/synth/test/file_global_queue_pause.html
Normal file
131
dom/media/webspeech/synth/test/file_global_queue_pause.html
Normal file
|
|
@ -0,0 +1,131 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=1188099
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 1188099: Calling pause() should work correctly with global queue</title>
|
||||
<script type="application/javascript">
|
||||
window.SimpleTest = parent.SimpleTest;
|
||||
window.info = parent.info;
|
||||
window.is = parent.is;
|
||||
window.isnot = parent.isnot;
|
||||
window.ok = parent.ok;
|
||||
window.todo = parent.todo;
|
||||
</script>
|
||||
<script type="application/javascript" src="common.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1188099">Mozilla Bug 1188099</a>
|
||||
<iframe id="frame1"></iframe>
|
||||
<iframe id="frame2"></iframe>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="application/javascript">
|
||||
Promise.all([loadFrame('frame1'), loadFrame('frame2')]).then(function ([frame1, frame2]) {
|
||||
var win1 = frame1.contentWindow;
|
||||
var win2 = frame2.contentWindow;
|
||||
|
||||
var utterance1 = new win1.SpeechSynthesisUtterance("Speak utterance 1.");
|
||||
utterance1.lang = 'it-IT-noend';
|
||||
var utterance2 = new win2.SpeechSynthesisUtterance("Speak utterance 2.");
|
||||
var utterance3 = new win1.SpeechSynthesisUtterance("Speak utterance 3.");
|
||||
var utterance4 = new win2.SpeechSynthesisUtterance("Speak utterance 4.");
|
||||
var eventOrder = ['start1', 'pause1', 'resume1', 'end1', 'start2', 'end2',
|
||||
'start4', 'end4', 'start3', 'end3'];
|
||||
|
||||
utterance1.addEventListener('start', function(e) {
|
||||
is(eventOrder.shift(), 'start1', 'start1');
|
||||
win1.speechSynthesis.pause();
|
||||
});
|
||||
utterance1.addEventListener('pause', function(e) {
|
||||
var expectedEvent = eventOrder.shift()
|
||||
is(expectedEvent, 'pause1', 'pause1');
|
||||
testSynthState(win1, { speaking: true, pending: false, paused: true});
|
||||
testSynthState(win2, { speaking: true, pending: true, paused: false});
|
||||
|
||||
if (expectedEvent == 'pause1') {
|
||||
win1.speechSynthesis.resume();
|
||||
}
|
||||
});
|
||||
utterance1.addEventListener('resume', function(e) {
|
||||
is(eventOrder.shift(), 'resume1', 'resume1');
|
||||
testSynthState(win1, { speaking: true, pending: false, paused: false});
|
||||
testSynthState(win2, { speaking: true, pending: true, paused: false});
|
||||
|
||||
win2.speechSynthesis.pause();
|
||||
|
||||
testSynthState(win1, { speaking: true, pending: false, paused: false});
|
||||
// 1188099: currently, paused state is not gaurenteed to be immediate.
|
||||
testSynthState(win2, { speaking: true, pending: true });
|
||||
|
||||
// We now make the utterance end.
|
||||
SpecialPowers.wrap(win1.speechSynthesis).forceEnd();
|
||||
});
|
||||
utterance1.addEventListener('end', function(e) {
|
||||
is(eventOrder.shift(), 'end1', 'end1');
|
||||
testSynthState(win1, { speaking: false, pending: false, paused: false});
|
||||
testSynthState(win2, { speaking: false, pending: true, paused: true});
|
||||
|
||||
win2.speechSynthesis.resume();
|
||||
});
|
||||
|
||||
utterance2.addEventListener('start', function(e) {
|
||||
is(eventOrder.shift(), 'start2', 'start2');
|
||||
testSynthState(win1, { speaking: true, pending: false, paused: false});
|
||||
testSynthState(win2, { speaking: true, pending: false, paused: false});
|
||||
});
|
||||
utterance2.addEventListener('end', function(e) {
|
||||
is(eventOrder.shift(), 'end2', 'end2');
|
||||
testSynthState(win1, { speaking: false, pending: false, paused: false});
|
||||
testSynthState(win2, { speaking: false, pending: false, paused: false});
|
||||
|
||||
win1.speechSynthesis.pause();
|
||||
|
||||
testSynthState(win1, { speaking: false, pending: false, paused: true});
|
||||
testSynthState(win2, { speaking: false, pending: false, paused: false});
|
||||
|
||||
win1.speechSynthesis.speak(utterance3);
|
||||
win2.speechSynthesis.speak(utterance4);
|
||||
|
||||
testSynthState(win1, { speaking: false, pending: true, paused: true});
|
||||
testSynthState(win2, { speaking: false, pending: true, paused: false});
|
||||
});
|
||||
|
||||
utterance4.addEventListener('start', function(e) {
|
||||
is(eventOrder.shift(), 'start4', 'start4');
|
||||
testSynthState(win1, { speaking: true, pending: true, paused: true});
|
||||
testSynthState(win2, { speaking: true, pending: false, paused: false});
|
||||
|
||||
win1.speechSynthesis.resume();
|
||||
});
|
||||
utterance4.addEventListener('end', function(e) {
|
||||
is(eventOrder.shift(), 'end4', 'end4');
|
||||
testSynthState(win1, { speaking: false, pending: true, paused: false});
|
||||
testSynthState(win2, { speaking: false, pending: false, paused: false});
|
||||
});
|
||||
|
||||
utterance3.addEventListener('start', function(e) {
|
||||
is(eventOrder.shift(), 'start3', 'start3');
|
||||
testSynthState(win1, { speaking: true, pending: false, paused: false});
|
||||
testSynthState(win2, { speaking: true, pending: false, paused: false});
|
||||
});
|
||||
|
||||
utterance3.addEventListener('end', function(e) {
|
||||
is(eventOrder.shift(), 'end3', 'end3');
|
||||
testSynthState(win1, { speaking: false, pending: false, paused: false});
|
||||
testSynthState(win2, { speaking: false, pending: false, paused: false});
|
||||
|
||||
SimpleTest.finish();
|
||||
});
|
||||
|
||||
win1.speechSynthesis.speak(utterance1);
|
||||
win2.speechSynthesis.speak(utterance2);
|
||||
});
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
102
dom/media/webspeech/synth/test/file_indirect_service_events.html
Normal file
102
dom/media/webspeech/synth/test/file_indirect_service_events.html
Normal file
|
|
@ -0,0 +1,102 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=1155034
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 1155034: Check that indirect audio services dispatch their own events</title>
|
||||
<script type="application/javascript">
|
||||
window.SimpleTest = parent.SimpleTest;
|
||||
window.info = parent.info;
|
||||
window.is = parent.is;
|
||||
window.isnot = parent.isnot;
|
||||
window.ok = parent.ok;
|
||||
</script>
|
||||
<script type="application/javascript" src="common.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1155034">Mozilla Bug 1155034</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="application/javascript">
|
||||
|
||||
/** Test for Bug 1155034 **/
|
||||
|
||||
function testFunc(done_cb) {
|
||||
function test_with_events() {
|
||||
info('test_with_events');
|
||||
var utterance = new SpeechSynthesisUtterance("never end, callback events");
|
||||
utterance.lang = 'it-IT-noend';
|
||||
|
||||
utterance.addEventListener('start', function(e) {
|
||||
info('start test_with_events');
|
||||
speechSynthesis.pause();
|
||||
// Wait to see if we get some bad events we didn't expect.
|
||||
});
|
||||
|
||||
utterance.addEventListener('pause', function(e) {
|
||||
is(e.charIndex, 1, 'pause event charIndex matches service arguments');
|
||||
is(e.elapsedTime, 1.5, 'pause event elapsedTime matches service arguments');
|
||||
speechSynthesis.resume();
|
||||
});
|
||||
|
||||
utterance.addEventListener('resume', function(e) {
|
||||
is(e.charIndex, 1, 'resume event charIndex matches service arguments');
|
||||
is(e.elapsedTime, 1.5, 'resume event elapsedTime matches service arguments');
|
||||
speechSynthesis.cancel();
|
||||
});
|
||||
|
||||
utterance.addEventListener('end', function(e) {
|
||||
ok(e.charIndex, 1, 'resume event charIndex matches service arguments');
|
||||
ok(e.elapsedTime, 1.5, 'end event elapsedTime matches service arguments');
|
||||
test_no_events();
|
||||
});
|
||||
|
||||
info('start speak');
|
||||
speechSynthesis.speak(utterance);
|
||||
}
|
||||
|
||||
function forbiddenEvent(e) {
|
||||
ok(false, 'no "' + e.type + '" event was explicitly dispatched from the service')
|
||||
}
|
||||
|
||||
function test_no_events() {
|
||||
info('test_no_events');
|
||||
var utterance = new SpeechSynthesisUtterance("never end");
|
||||
utterance.lang = "it-IT-noevents-noend";
|
||||
utterance.addEventListener('start', function(e) {
|
||||
speechSynthesis.pause();
|
||||
// Wait to see if we get some bad events we didn't expect.
|
||||
setTimeout(function() {
|
||||
ok(true, 'didn\'t get any unwanted events');
|
||||
utterance.removeEventListener('end', forbiddenEvent);
|
||||
SpecialPowers.wrap(speechSynthesis).forceEnd();
|
||||
done_cb();
|
||||
}, 1000);
|
||||
});
|
||||
|
||||
utterance.addEventListener('pause', forbiddenEvent);
|
||||
utterance.addEventListener('end', forbiddenEvent);
|
||||
|
||||
speechSynthesis.speak(utterance);
|
||||
}
|
||||
|
||||
test_with_events();
|
||||
}
|
||||
|
||||
// Run test with no global queue, and then run it with a global queue.
|
||||
testFunc(function() {
|
||||
SpecialPowers.pushPrefEnv(
|
||||
{ set: [['media.webspeech.synth.force_global_queue', true]] }, function() {
|
||||
testFunc(SimpleTest.finish)
|
||||
});
|
||||
});
|
||||
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
95
dom/media/webspeech/synth/test/file_setup.html
Normal file
95
dom/media/webspeech/synth/test/file_setup.html
Normal file
|
|
@ -0,0 +1,95 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=525444
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 525444: Web Speech API check all classes are present</title>
|
||||
<script type="application/javascript">
|
||||
window.SimpleTest = parent.SimpleTest;
|
||||
window.is = parent.is;
|
||||
window.isnot = parent.isnot;
|
||||
window.ok = parent.ok;
|
||||
</script>
|
||||
<script type="application/javascript" src="common.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="application/javascript">
|
||||
|
||||
/** Test for Bug 525444 **/
|
||||
|
||||
ok(SpeechSynthesis, "SpeechSynthesis exists in global scope");
|
||||
ok(SpeechSynthesisVoice, "SpeechSynthesisVoice exists in global scope");
|
||||
ok(SpeechSynthesisErrorEvent, "SpeechSynthesisErrorEvent exists in global scope");
|
||||
ok(SpeechSynthesisEvent, "SpeechSynthesisEvent exists in global scope");
|
||||
|
||||
// SpeechSynthesisUtterance is the only type that has a constructor
|
||||
// and writable properties
|
||||
ok(SpeechSynthesisUtterance, "SpeechSynthesisUtterance exists in global scope");
|
||||
var ssu = new SpeechSynthesisUtterance("hello world");
|
||||
is(typeof ssu, "object", "SpeechSynthesisUtterance instance is an object");
|
||||
is(ssu.text, "hello world", "SpeechSynthesisUtterance.text is correct");
|
||||
is(ssu.volume, 1, "SpeechSynthesisUtterance.volume default is correct");
|
||||
is(ssu.rate, 1, "SpeechSynthesisUtterance.rate default is correct");
|
||||
is(ssu.pitch, 1, "SpeechSynthesisUtterance.pitch default is correct");
|
||||
ssu.lang = "he-IL";
|
||||
ssu.volume = 0.5;
|
||||
ssu.rate = 2.0;
|
||||
ssu.pitch = 1.5;
|
||||
is(ssu.lang, "he-IL", "SpeechSynthesisUtterance.lang is correct");
|
||||
is(ssu.volume, 0.5, "SpeechSynthesisUtterance.volume is correct");
|
||||
is(ssu.rate, 2.0, "SpeechSynthesisUtterance.rate is correct");
|
||||
is(ssu.pitch, 1.5, "SpeechSynthesisUtterance.pitch is correct");
|
||||
|
||||
// Assign a rate that is out of bounds
|
||||
ssu.rate = 20;
|
||||
is(ssu.rate, 10, "SpeechSynthesisUtterance.rate enforces max of 10");
|
||||
ssu.rate = 0;
|
||||
is(ssu.rate.toPrecision(1), "0.1", "SpeechSynthesisUtterance.rate enforces min of 0.1");
|
||||
|
||||
// Assign a volume which is out of bounds
|
||||
ssu.volume = 2;
|
||||
is(ssu.volume, 1, "SpeechSynthesisUtterance.volume enforces max of 1");
|
||||
ssu.volume = -1;
|
||||
is(ssu.volume, 0, "SpeechSynthesisUtterance.volume enforces min of 0");
|
||||
|
||||
// Assign a pitch which is out of bounds
|
||||
ssu.pitch = 2.1;
|
||||
is(ssu.pitch, 2, "SpeechSynthesisUtterance.pitch enforces max of 2");
|
||||
ssu.pitch = -1;
|
||||
is(ssu.pitch, 0, "SpeechSynthesisUtterance.pitch enforces min of 0");
|
||||
|
||||
// Test for singleton instance hanging off of window.
|
||||
ok(speechSynthesis, "speechSynthesis exists in global scope");
|
||||
is(typeof speechSynthesis, "object", "speechSynthesis instance is an object");
|
||||
is(typeof speechSynthesis.speak, "function", "speechSynthesis.speak is a function");
|
||||
is(typeof speechSynthesis.cancel, "function", "speechSynthesis.cancel is a function");
|
||||
is(typeof speechSynthesis.pause, "function", "speechSynthesis.pause is a function");
|
||||
is(typeof speechSynthesis.resume, "function", "speechSynthesis.resume is a function");
|
||||
is(typeof speechSynthesis.getVoices, "function", "speechSynthesis.getVoices is a function");
|
||||
|
||||
is(typeof speechSynthesis.pending, "boolean", "speechSynthesis.pending is a boolean");
|
||||
is(typeof speechSynthesis.speaking, "boolean", "speechSynthesis.speaking is a boolean");
|
||||
is(typeof speechSynthesis.paused, "boolean", "speechSynthesis.paused is a boolean");
|
||||
|
||||
var voices1 = speechSynthesis.getVoices();
|
||||
var voices2 = speechSynthesis.getVoices();
|
||||
|
||||
ok(voices1.length == voices2.length, "Voice count matches");
|
||||
|
||||
for (var i in voices1) {
|
||||
ok(voices1[i] == voices2[i], "Voice instance matches");
|
||||
}
|
||||
|
||||
SimpleTest.finish();
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
100
dom/media/webspeech/synth/test/file_speech_cancel.html
Normal file
100
dom/media/webspeech/synth/test/file_speech_cancel.html
Normal file
|
|
@ -0,0 +1,100 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=1150315
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 1150315: Check that successive cancel/speak calls work</title>
|
||||
<script type="application/javascript">
|
||||
window.SimpleTest = parent.SimpleTest;
|
||||
window.info = parent.info;
|
||||
window.is = parent.is;
|
||||
window.isnot = parent.isnot;
|
||||
window.ok = parent.ok;
|
||||
</script>
|
||||
<script type="application/javascript" src="common.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1150315">Mozilla Bug 1150315</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="application/javascript">
|
||||
|
||||
/** Test for Bug 1150315 **/
|
||||
|
||||
function testFunc(done_cb) {
|
||||
var gotEndEvent = false;
|
||||
// A long utterance that we will interrupt.
|
||||
var utterance = new SpeechSynthesisUtterance("Donec ac nunc feugiat, posuere " +
|
||||
"mauris id, pharetra velit. Donec fermentum orci nunc, sit amet maximus" +
|
||||
"dui tincidunt ut. Sed ultricies ac nisi a laoreet. Proin interdum," +
|
||||
"libero maximus hendrerit posuere, lorem risus egestas nisl, a" +
|
||||
"ultricies massa justo eu nisi. Duis mattis nibh a ligula tincidunt" +
|
||||
"tincidunt non eu erat. Sed bibendum varius vulputate. Cras leo magna," +
|
||||
"ornare ac posuere vel, luctus id metus. Mauris nec quam ac augue" +
|
||||
"consectetur bibendum. Integer a commodo tortor. Duis semper dolor eu" +
|
||||
"facilisis facilisis. Etiam venenatis turpis est, quis tincidunt velit" +
|
||||
"suscipit a. Cras semper orci in sapien rhoncus bibendum. Suspendisse" +
|
||||
"eu ex lobortis, finibus enim in, condimentum quam. Maecenas eget dui" +
|
||||
"ipsum. Aliquam tortor leo, interdum eget congue ut, tempor id elit.");
|
||||
utterance.addEventListener('start', function(e) {
|
||||
ok(true, 'start utterance 1');
|
||||
speechSynthesis.cancel();
|
||||
info('cancel!');
|
||||
speechSynthesis.speak(utterance2);
|
||||
info('speak??');
|
||||
});
|
||||
|
||||
var utterance2 = new SpeechSynthesisUtterance("Proin ornare neque vitae " +
|
||||
"risus mattis rutrum. Suspendisse a velit ut est convallis aliquet." +
|
||||
"Nullam ante elit, malesuada vel luctus rutrum, ultricies nec libero." +
|
||||
"Praesent eu iaculis orci. Sed nisl diam, sodales ac purus et," +
|
||||
"volutpat interdum tortor. Nullam aliquam porta elit et maximus. Cras" +
|
||||
"risus lectus, elementum vel sodales vel, ultricies eget lectus." +
|
||||
"Curabitur velit lacus, mollis vel finibus et, molestie sit amet" +
|
||||
"sapien. Proin vitae dolor ac augue posuere efficitur ac scelerisque" +
|
||||
"diam. Nulla sed odio elit.");
|
||||
utterance2.addEventListener('start', function() {
|
||||
info('start');
|
||||
speechSynthesis.cancel();
|
||||
speechSynthesis.speak(utterance3);
|
||||
});
|
||||
utterance2.addEventListener('end', function(e) {
|
||||
gotEndEvent = true;
|
||||
});
|
||||
|
||||
var utterance3 = new SpeechSynthesisUtterance("Hello, world 3!");
|
||||
utterance3.addEventListener('start', function() {
|
||||
ok(gotEndEvent, "didn't get start event for this utterance");
|
||||
});
|
||||
utterance3.addEventListener('end', done_cb);
|
||||
|
||||
// Speak/cancel while paused (Bug 1187105)
|
||||
speechSynthesis.pause();
|
||||
speechSynthesis.speak(new SpeechSynthesisUtterance("hello."));
|
||||
ok(speechSynthesis.pending, "paused speechSynthesis has an utterance queued.");
|
||||
speechSynthesis.cancel();
|
||||
ok(!speechSynthesis.pending, "paused speechSynthesis has no utterance queued.");
|
||||
speechSynthesis.resume();
|
||||
|
||||
speechSynthesis.speak(utterance);
|
||||
ok(!speechSynthesis.speaking, "speechSynthesis is not speaking yet.");
|
||||
ok(speechSynthesis.pending, "speechSynthesis has an utterance queued.");
|
||||
}
|
||||
|
||||
// Run test with no global queue, and then run it with a global queue.
|
||||
testFunc(function() {
|
||||
SpecialPowers.pushPrefEnv(
|
||||
{ set: [['media.webspeech.synth.force_global_queue', true]] }, function() {
|
||||
testFunc(SimpleTest.finish)
|
||||
});
|
||||
});
|
||||
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
46
dom/media/webspeech/synth/test/file_speech_error.html
Normal file
46
dom/media/webspeech/synth/test/file_speech_error.html
Normal file
|
|
@ -0,0 +1,46 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=1226015
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 1226015</title>
|
||||
<script type="application/javascript">
|
||||
window.SimpleTest = parent.SimpleTest;
|
||||
window.info = parent.info;
|
||||
window.is = parent.is;
|
||||
window.isnot = parent.isnot;
|
||||
window.ok = parent.ok;
|
||||
</script>
|
||||
<script type="application/javascript" src="common.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1226015">Mozilla Bug 1226015</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="application/javascript">
|
||||
|
||||
/** Test for Bug 1226015 **/
|
||||
|
||||
function testFunc(done_cb) {
|
||||
var utterance = new SpeechSynthesisUtterance();
|
||||
utterance.lang = 'it-IT-failatstart';
|
||||
|
||||
speechSynthesis.speak(utterance);
|
||||
speechSynthesis.cancel();
|
||||
|
||||
ok(true, "we didn't crash, that is good.")
|
||||
SimpleTest.finish();
|
||||
}
|
||||
|
||||
// Run test with no global queue, and then run it with a global queue.
|
||||
testFunc();
|
||||
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
85
dom/media/webspeech/synth/test/file_speech_queue.html
Normal file
85
dom/media/webspeech/synth/test/file_speech_queue.html
Normal file
|
|
@ -0,0 +1,85 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html lang="en-US">
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=525444
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 525444: Web Speech API, check speech synth queue</title>
|
||||
<script type="application/javascript">
|
||||
window.SimpleTest = parent.SimpleTest;
|
||||
window.is = parent.is;
|
||||
window.isnot = parent.isnot;
|
||||
window.ok = parent.ok;
|
||||
</script>
|
||||
<script type="application/javascript" src="common.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=525444">Mozilla Bug 525444</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="application/javascript">
|
||||
|
||||
/** Test for Bug 525444 **/
|
||||
|
||||
// XXX: Rate and pitch are not tested.
|
||||
|
||||
var langUriMap = {};
|
||||
|
||||
for (var voice of speechSynthesis.getVoices()) {
|
||||
langUriMap[voice.lang] = voice.voiceURI;
|
||||
ok(true, voice.lang + ' ' + voice.voiceURI + ' ' + voice.default);
|
||||
is(voice.default, voice.lang == 'en-JM', 'Only Jamaican voice should be default');
|
||||
}
|
||||
|
||||
ok(langUriMap['en-JM'], 'No English-Jamaican voice');
|
||||
ok(langUriMap['en-GB'], 'No English-British voice');
|
||||
ok(langUriMap['en-CA'], 'No English-Canadian voice');
|
||||
ok(langUriMap['fr-CA'], 'No French-Canadian voice');
|
||||
ok(langUriMap['es-MX'], 'No Spanish-Mexican voice');
|
||||
ok(langUriMap['it-IT-fail'], 'No Failing Italian voice');
|
||||
|
||||
function testFunc(done_cb) {
|
||||
synthTestQueue(
|
||||
[[{text: "Hello, world."},
|
||||
{ uri: langUriMap['en-JM'] }],
|
||||
[{text: "Bonjour tout le monde .",
|
||||
args: { lang: "fr", rate: 0.5, pitch: 0.75 }},
|
||||
{ uri: langUriMap['fr-CA'], rate: 0.5, pitch: 0.75}],
|
||||
[{text: "How are you doing?", args: { lang: "en-GB" } },
|
||||
{ rate: 1, pitch: 1, uri: langUriMap['en-GB']}],
|
||||
[{text: "Come stai?", args: { lang: "it-IT-fail" } },
|
||||
{ rate: 1, pitch: 1, uri: langUriMap['it-IT-fail'], err: true }],
|
||||
[{text: "¡hasta mañana!", args: { lang: "es-MX" } },
|
||||
{ uri: langUriMap['es-MX'] }]],
|
||||
function () {
|
||||
var test_data = [];
|
||||
var voices = speechSynthesis.getVoices();
|
||||
for (var voice of voices) {
|
||||
if (voice.voiceURI.indexOf('urn:moz-tts:fake-direct') < 0) {
|
||||
continue;
|
||||
}
|
||||
test_data.push([{text: "Hello world", args: { voice: voice} },
|
||||
{uri: voice.voiceURI}]);
|
||||
}
|
||||
|
||||
synthTestQueue(test_data, done_cb);
|
||||
});
|
||||
}
|
||||
|
||||
// Run test with no global queue, and then run it with a global queue.
|
||||
testFunc(function() {
|
||||
SpecialPowers.pushPrefEnv(
|
||||
{ set: [['media.webspeech.synth.force_global_queue', true]] }, function() {
|
||||
testFunc(SimpleTest.finish)
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
53
dom/media/webspeech/synth/test/file_speech_simple.html
Normal file
53
dom/media/webspeech/synth/test/file_speech_simple.html
Normal file
|
|
@ -0,0 +1,53 @@
|
|||
<!DOCTYPE HTML>
|
||||
<html>
|
||||
<!--
|
||||
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
|
||||
-->
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>Test for Bug 650295: Web Speech API check all classes are present</title>
|
||||
<script type="application/javascript">
|
||||
window.SimpleTest = parent.SimpleTest;
|
||||
window.info = parent.info;
|
||||
window.is = parent.is;
|
||||
window.isnot = parent.isnot;
|
||||
window.ok = parent.ok;
|
||||
</script>
|
||||
<script type="application/javascript" src="common.js"></script>
|
||||
</head>
|
||||
<body>
|
||||
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
|
||||
<p id="display"></p>
|
||||
<div id="content" style="display: none">
|
||||
|
||||
</div>
|
||||
<pre id="test">
|
||||
<script type="application/javascript">
|
||||
|
||||
/** Test for Bug 525444 **/
|
||||
|
||||
var gotStartEvent = false;
|
||||
var gotBoundaryEvent = false;
|
||||
var utterance = new SpeechSynthesisUtterance("Hello, world!");
|
||||
utterance.addEventListener('start', function(e) {
|
||||
ok(speechSynthesis.speaking, "speechSynthesis is speaking.");
|
||||
ok(!speechSynthesis.pending, "speechSynthesis has no other utterances queued.");
|
||||
gotStartEvent = true;
|
||||
});
|
||||
|
||||
utterance.addEventListener('end', function(e) {
|
||||
ok(!speechSynthesis.speaking, "speechSynthesis is not speaking.");
|
||||
ok(!speechSynthesis.pending, "speechSynthesis has no other utterances queued.");
|
||||
ok(gotStartEvent, "Got 'start' event.");
|
||||
info('end ' + e.elapsedTime);
|
||||
SimpleTest.finish();
|
||||
});
|
||||
|
||||
speechSynthesis.speak(utterance);
|
||||
ok(!speechSynthesis.speaking, "speechSynthesis is not speaking yet.");
|
||||
ok(speechSynthesis.pending, "speechSynthesis has an utterance queued.");
|
||||
|
||||
</script>
|
||||
</pre>
|
||||
</body>
|
||||
</html>
|
||||
26
dom/media/webspeech/synth/test/mochitest.ini
Normal file
26
dom/media/webspeech/synth/test/mochitest.ini
Normal file
|
|
@ -0,0 +1,26 @@
|
|||
[DEFAULT]
|
||||
tags=msg
|
||||
subsuite = media
|
||||
support-files =
|
||||
common.js
|
||||
file_bfcache_frame.html
|
||||
file_setup.html
|
||||
file_speech_queue.html
|
||||
file_speech_simple.html
|
||||
file_speech_cancel.html
|
||||
file_speech_error.html
|
||||
file_indirect_service_events.html
|
||||
file_global_queue.html
|
||||
file_global_queue_cancel.html
|
||||
file_global_queue_pause.html
|
||||
|
||||
[test_setup.html]
|
||||
[test_speech_queue.html]
|
||||
[test_speech_simple.html]
|
||||
[test_speech_cancel.html]
|
||||
[test_speech_error.html]
|
||||
[test_indirect_service_events.html]
|
||||
[test_global_queue.html]
|
||||
[test_global_queue_cancel.html]
|
||||
[test_global_queue_pause.html]
|
||||
[test_bfcache.html]
|
||||
401
dom/media/webspeech/synth/test/nsFakeSynthServices.cpp
Normal file
401
dom/media/webspeech/synth/test/nsFakeSynthServices.cpp
Normal file
|
|
@ -0,0 +1,401 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#include "nsISupports.h"
|
||||
#include "nsFakeSynthServices.h"
|
||||
#include "nsPrintfCString.h"
|
||||
#include "nsIWeakReferenceUtils.h"
|
||||
#include "SharedBuffer.h"
|
||||
#include "nsISimpleEnumerator.h"
|
||||
|
||||
#include "mozilla/dom/nsSynthVoiceRegistry.h"
|
||||
#include "mozilla/dom/nsSpeechTask.h"
|
||||
|
||||
#include "nsThreadUtils.h"
|
||||
#include "prenv.h"
|
||||
#include "mozilla/Preferences.h"
|
||||
#include "mozilla/DebugOnly.h"
|
||||
|
||||
#define CHANNELS 1
|
||||
#define SAMPLERATE 1600
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
StaticRefPtr<nsFakeSynthServices> nsFakeSynthServices::sSingleton;
|
||||
|
||||
enum VoiceFlags
|
||||
{
|
||||
eSuppressEvents = 1,
|
||||
eSuppressEnd = 2,
|
||||
eFailAtStart = 4,
|
||||
eFail = 8
|
||||
};
|
||||
|
||||
struct VoiceDetails
|
||||
{
|
||||
const char* uri;
|
||||
const char* name;
|
||||
const char* lang;
|
||||
bool defaultVoice;
|
||||
uint32_t flags;
|
||||
};
|
||||
|
||||
static const VoiceDetails sDirectVoices[] = {
|
||||
{"urn:moz-tts:fake-direct:bob", "Bob Marley", "en-JM", true, 0},
|
||||
{"urn:moz-tts:fake-direct:amy", "Amy Winehouse", "en-GB", false, 0},
|
||||
{"urn:moz-tts:fake-direct:lenny", "Leonard Cohen", "en-CA", false, 0},
|
||||
{"urn:moz-tts:fake-direct:celine", "Celine Dion", "fr-CA", false, 0},
|
||||
{"urn:moz-tts:fake-direct:julie", "Julieta Venegas", "es-MX", false, },
|
||||
};
|
||||
|
||||
static const VoiceDetails sIndirectVoices[] = {
|
||||
{"urn:moz-tts:fake-indirect:zanetta", "Zanetta Farussi", "it-IT", false, 0},
|
||||
{"urn:moz-tts:fake-indirect:margherita", "Margherita Durastanti", "it-IT-noevents-noend", false, eSuppressEvents | eSuppressEnd},
|
||||
{"urn:moz-tts:fake-indirect:teresa", "Teresa Cornelys", "it-IT-noend", false, eSuppressEnd},
|
||||
{"urn:moz-tts:fake-indirect:cecilia", "Cecilia Bartoli", "it-IT-failatstart", false, eFailAtStart},
|
||||
{"urn:moz-tts:fake-indirect:gottardo", "Gottardo Aldighieri", "it-IT-fail", false, eFail},
|
||||
};
|
||||
|
||||
// FakeSynthCallback
|
||||
class FakeSynthCallback : public nsISpeechTaskCallback
|
||||
{
|
||||
public:
|
||||
explicit FakeSynthCallback(nsISpeechTask* aTask) : mTask(aTask) { }
|
||||
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
|
||||
NS_DECL_CYCLE_COLLECTION_CLASS_AMBIGUOUS(FakeSynthCallback, nsISpeechTaskCallback)
|
||||
|
||||
NS_IMETHOD OnPause() override
|
||||
{
|
||||
if (mTask) {
|
||||
mTask->DispatchPause(1.5, 1);
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHOD OnResume() override
|
||||
{
|
||||
if (mTask) {
|
||||
mTask->DispatchResume(1.5, 1);
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHOD OnCancel() override
|
||||
{
|
||||
if (mTask) {
|
||||
mTask->DispatchEnd(1.5, 1);
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHOD OnVolumeChanged(float aVolume) override
|
||||
{
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
private:
|
||||
virtual ~FakeSynthCallback() { }
|
||||
|
||||
nsCOMPtr<nsISpeechTask> mTask;
|
||||
};
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTION(FakeSynthCallback, mTask);
|
||||
|
||||
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(FakeSynthCallback)
|
||||
NS_INTERFACE_MAP_ENTRY(nsISpeechTaskCallback)
|
||||
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsISpeechTaskCallback)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
NS_IMPL_CYCLE_COLLECTING_ADDREF(FakeSynthCallback)
|
||||
NS_IMPL_CYCLE_COLLECTING_RELEASE(FakeSynthCallback)
|
||||
|
||||
// FakeDirectAudioSynth
|
||||
|
||||
class FakeDirectAudioSynth : public nsISpeechService
|
||||
{
|
||||
|
||||
public:
|
||||
FakeDirectAudioSynth() { }
|
||||
|
||||
NS_DECL_ISUPPORTS
|
||||
NS_DECL_NSISPEECHSERVICE
|
||||
|
||||
private:
|
||||
virtual ~FakeDirectAudioSynth() { }
|
||||
};
|
||||
|
||||
NS_IMPL_ISUPPORTS(FakeDirectAudioSynth, nsISpeechService)
|
||||
|
||||
NS_IMETHODIMP
|
||||
FakeDirectAudioSynth::Speak(const nsAString& aText, const nsAString& aUri,
|
||||
float aVolume, float aRate, float aPitch,
|
||||
nsISpeechTask* aTask)
|
||||
{
|
||||
class Runnable final : public mozilla::Runnable
|
||||
{
|
||||
public:
|
||||
Runnable(nsISpeechTask* aTask, const nsAString& aText) :
|
||||
mTask(aTask), mText(aText)
|
||||
{
|
||||
}
|
||||
|
||||
NS_IMETHOD Run() override
|
||||
{
|
||||
RefPtr<FakeSynthCallback> cb = new FakeSynthCallback(nullptr);
|
||||
mTask->Setup(cb, CHANNELS, SAMPLERATE, 2);
|
||||
|
||||
// Just an arbitrary multiplier. Pretend that each character is
|
||||
// synthesized to 40 frames.
|
||||
uint32_t frames_length = 40 * mText.Length();
|
||||
auto frames = MakeUnique<int16_t[]>(frames_length);
|
||||
mTask->SendAudioNative(frames.get(), frames_length);
|
||||
|
||||
mTask->SendAudioNative(nullptr, 0);
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
private:
|
||||
nsCOMPtr<nsISpeechTask> mTask;
|
||||
nsString mText;
|
||||
};
|
||||
|
||||
nsCOMPtr<nsIRunnable> runnable = new Runnable(aTask, aText);
|
||||
NS_DispatchToMainThread(runnable);
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
FakeDirectAudioSynth::GetServiceType(SpeechServiceType* aServiceType)
|
||||
{
|
||||
*aServiceType = nsISpeechService::SERVICETYPE_DIRECT_AUDIO;
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
// FakeDirectAudioSynth
|
||||
|
||||
class FakeIndirectAudioSynth : public nsISpeechService
|
||||
{
|
||||
|
||||
public:
|
||||
FakeIndirectAudioSynth() {}
|
||||
|
||||
NS_DECL_ISUPPORTS
|
||||
NS_DECL_NSISPEECHSERVICE
|
||||
|
||||
private:
|
||||
virtual ~FakeIndirectAudioSynth() { }
|
||||
};
|
||||
|
||||
NS_IMPL_ISUPPORTS(FakeIndirectAudioSynth, nsISpeechService)
|
||||
|
||||
NS_IMETHODIMP
|
||||
FakeIndirectAudioSynth::Speak(const nsAString& aText, const nsAString& aUri,
|
||||
float aVolume, float aRate, float aPitch,
|
||||
nsISpeechTask* aTask)
|
||||
{
|
||||
class DispatchStart final : public Runnable
|
||||
{
|
||||
public:
|
||||
explicit DispatchStart(nsISpeechTask* aTask) :
|
||||
mTask(aTask)
|
||||
{
|
||||
}
|
||||
|
||||
NS_IMETHOD Run() override
|
||||
{
|
||||
mTask->DispatchStart();
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
private:
|
||||
nsCOMPtr<nsISpeechTask> mTask;
|
||||
};
|
||||
|
||||
class DispatchEnd final : public Runnable
|
||||
{
|
||||
public:
|
||||
DispatchEnd(nsISpeechTask* aTask, const nsAString& aText) :
|
||||
mTask(aTask), mText(aText)
|
||||
{
|
||||
}
|
||||
|
||||
NS_IMETHOD Run() override
|
||||
{
|
||||
mTask->DispatchEnd(mText.Length()/2, mText.Length());
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
private:
|
||||
nsCOMPtr<nsISpeechTask> mTask;
|
||||
nsString mText;
|
||||
};
|
||||
|
||||
class DispatchError final : public Runnable
|
||||
{
|
||||
public:
|
||||
DispatchError(nsISpeechTask* aTask, const nsAString& aText) :
|
||||
mTask(aTask), mText(aText)
|
||||
{
|
||||
}
|
||||
|
||||
NS_IMETHOD Run() override
|
||||
{
|
||||
mTask->DispatchError(mText.Length()/2, mText.Length());
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
private:
|
||||
nsCOMPtr<nsISpeechTask> mTask;
|
||||
nsString mText;
|
||||
};
|
||||
|
||||
uint32_t flags = 0;
|
||||
for (uint32_t i = 0; i < ArrayLength(sIndirectVoices); i++) {
|
||||
if (aUri.EqualsASCII(sIndirectVoices[i].uri)) {
|
||||
flags = sIndirectVoices[i].flags;
|
||||
}
|
||||
}
|
||||
|
||||
if (flags & eFailAtStart) {
|
||||
return NS_ERROR_FAILURE;
|
||||
}
|
||||
|
||||
RefPtr<FakeSynthCallback> cb = new FakeSynthCallback(
|
||||
(flags & eSuppressEvents) ? nullptr : aTask);
|
||||
|
||||
aTask->Setup(cb, 0, 0, 0);
|
||||
|
||||
nsCOMPtr<nsIRunnable> runnable = new DispatchStart(aTask);
|
||||
NS_DispatchToMainThread(runnable);
|
||||
|
||||
if (flags & eFail) {
|
||||
runnable = new DispatchError(aTask, aText);
|
||||
NS_DispatchToMainThread(runnable);
|
||||
} else if ((flags & eSuppressEnd) == 0) {
|
||||
runnable = new DispatchEnd(aTask, aText);
|
||||
NS_DispatchToMainThread(runnable);
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
NS_IMETHODIMP
|
||||
FakeIndirectAudioSynth::GetServiceType(SpeechServiceType* aServiceType)
|
||||
{
|
||||
*aServiceType = nsISpeechService::SERVICETYPE_INDIRECT_AUDIO;
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
// nsFakeSynthService
|
||||
|
||||
NS_INTERFACE_MAP_BEGIN(nsFakeSynthServices)
|
||||
NS_INTERFACE_MAP_ENTRY(nsIObserver)
|
||||
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsIObserver)
|
||||
NS_INTERFACE_MAP_END
|
||||
|
||||
NS_IMPL_ADDREF(nsFakeSynthServices)
|
||||
NS_IMPL_RELEASE(nsFakeSynthServices)
|
||||
|
||||
nsFakeSynthServices::nsFakeSynthServices()
|
||||
{
|
||||
}
|
||||
|
||||
nsFakeSynthServices::~nsFakeSynthServices()
|
||||
{
|
||||
}
|
||||
|
||||
static void
|
||||
AddVoices(nsISpeechService* aService, const VoiceDetails* aVoices, uint32_t aLength)
|
||||
{
|
||||
RefPtr<nsSynthVoiceRegistry> registry = nsSynthVoiceRegistry::GetInstance();
|
||||
for (uint32_t i = 0; i < aLength; i++) {
|
||||
NS_ConvertUTF8toUTF16 name(aVoices[i].name);
|
||||
NS_ConvertUTF8toUTF16 uri(aVoices[i].uri);
|
||||
NS_ConvertUTF8toUTF16 lang(aVoices[i].lang);
|
||||
// These services can handle more than one utterance at a time and have
|
||||
// several speaking simultaniously. So, aQueuesUtterances == false
|
||||
registry->AddVoice(aService, uri, name, lang, true, false);
|
||||
if (aVoices[i].defaultVoice) {
|
||||
registry->SetDefaultVoice(uri, true);
|
||||
}
|
||||
}
|
||||
|
||||
registry->NotifyVoicesChanged();
|
||||
}
|
||||
|
||||
void
|
||||
nsFakeSynthServices::Init()
|
||||
{
|
||||
mDirectService = new FakeDirectAudioSynth();
|
||||
AddVoices(mDirectService, sDirectVoices, ArrayLength(sDirectVoices));
|
||||
|
||||
mIndirectService = new FakeIndirectAudioSynth();
|
||||
AddVoices(mIndirectService, sIndirectVoices, ArrayLength(sIndirectVoices));
|
||||
}
|
||||
|
||||
// nsIObserver
|
||||
|
||||
NS_IMETHODIMP
|
||||
nsFakeSynthServices::Observe(nsISupports* aSubject, const char* aTopic,
|
||||
const char16_t* aData)
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
if(NS_WARN_IF(!(!strcmp(aTopic, "speech-synth-started")))) {
|
||||
return NS_ERROR_UNEXPECTED;
|
||||
}
|
||||
|
||||
if (Preferences::GetBool("media.webspeech.synth.test")) {
|
||||
NS_DispatchToMainThread(NewRunnableMethod(this, &nsFakeSynthServices::Init));
|
||||
}
|
||||
|
||||
return NS_OK;
|
||||
}
|
||||
|
||||
// static methods
|
||||
|
||||
nsFakeSynthServices*
|
||||
nsFakeSynthServices::GetInstance()
|
||||
{
|
||||
MOZ_ASSERT(NS_IsMainThread());
|
||||
if (!XRE_IsParentProcess()) {
|
||||
MOZ_ASSERT(false, "nsFakeSynthServices can only be started on main gecko process");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
if (!sSingleton) {
|
||||
sSingleton = new nsFakeSynthServices();
|
||||
}
|
||||
|
||||
return sSingleton;
|
||||
}
|
||||
|
||||
already_AddRefed<nsFakeSynthServices>
|
||||
nsFakeSynthServices::GetInstanceForService()
|
||||
{
|
||||
RefPtr<nsFakeSynthServices> picoService = GetInstance();
|
||||
return picoService.forget();
|
||||
}
|
||||
|
||||
void
|
||||
nsFakeSynthServices::Shutdown()
|
||||
{
|
||||
if (!sSingleton) {
|
||||
return;
|
||||
}
|
||||
|
||||
sSingleton = nullptr;
|
||||
}
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
52
dom/media/webspeech/synth/test/nsFakeSynthServices.h
Normal file
52
dom/media/webspeech/synth/test/nsFakeSynthServices.h
Normal file
|
|
@ -0,0 +1,52 @@
|
|||
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
|
||||
/* vim:set ts=2 sw=2 sts=2 et cindent: */
|
||||
/* This Source Code Form is subject to the terms of the Mozilla Public
|
||||
* License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
|
||||
|
||||
#ifndef nsFakeSynthServices_h
|
||||
#define nsFakeSynthServices_h
|
||||
|
||||
#include "nsTArray.h"
|
||||
#include "nsIObserver.h"
|
||||
#include "nsIThread.h"
|
||||
#include "nsISpeechService.h"
|
||||
#include "nsRefPtrHashtable.h"
|
||||
#include "mozilla/StaticPtr.h"
|
||||
#include "mozilla/Monitor.h"
|
||||
|
||||
namespace mozilla {
|
||||
namespace dom {
|
||||
|
||||
class nsFakeSynthServices : public nsIObserver
|
||||
{
|
||||
|
||||
public:
|
||||
NS_DECL_ISUPPORTS
|
||||
NS_DECL_NSIOBSERVER
|
||||
|
||||
nsFakeSynthServices();
|
||||
|
||||
static nsFakeSynthServices* GetInstance();
|
||||
|
||||
static already_AddRefed<nsFakeSynthServices> GetInstanceForService();
|
||||
|
||||
static void Shutdown();
|
||||
|
||||
private:
|
||||
|
||||
virtual ~nsFakeSynthServices();
|
||||
|
||||
void Init();
|
||||
|
||||
nsCOMPtr<nsISpeechService> mDirectService;
|
||||
|
||||
nsCOMPtr<nsISpeechService> mIndirectService;
|
||||
|
||||
static StaticRefPtr<nsFakeSynthServices> sSingleton;
|
||||
};
|
||||
|
||||
} // namespace dom
|
||||
} // namespace mozilla
|
||||
|
||||
#endif
|
||||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Add a link
Reference in a new issue