import FIREFOX_52_6_0esr_RELEASE from mozilla-esr52 hg repo

This commit is contained in:
Roy Tam 2018-01-19 03:59:58 +08:00
commit dcd9973243
150858 changed files with 23884658 additions and 0 deletions

View file

@ -0,0 +1,9 @@
# vim: set filetype=python:
# This Source Code Form is subject to the terms of the Mozilla Public
# License, v. 2.0. If a copy of the MPL was not distributed with this
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
DIRS = ['synth']
if CONFIG['MOZ_WEBSPEECH']:
DIRS += ['recognition']

View file

@ -0,0 +1,357 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsThreadUtils.h"
#include "nsXPCOMCIDInternal.h"
#include "PocketSphinxSpeechRecognitionService.h"
#include "nsIFile.h"
#include "SpeechGrammar.h"
#include "SpeechRecognition.h"
#include "SpeechRecognitionAlternative.h"
#include "SpeechRecognitionResult.h"
#include "SpeechRecognitionResultList.h"
#include "nsIObserverService.h"
#include "MediaPrefs.h"
#include "mozilla/Services.h"
#include "nsDirectoryServiceDefs.h"
#include "nsDirectoryServiceUtils.h"
#include "nsMemory.h"
extern "C" {
#include "pocketsphinx/pocketsphinx.h"
#include "sphinxbase/logmath.h"
#include "sphinxbase/sphinx_config.h"
#include "sphinxbase/jsgf.h"
}
namespace mozilla {
using namespace dom;
class DecodeResultTask : public Runnable
{
public:
DecodeResultTask(const nsString& hypstring,
float64 confidence,
WeakPtr<dom::SpeechRecognition> recognition)
: mResult(hypstring),
mConfidence(confidence),
mRecognition(recognition),
mWorkerThread(do_GetCurrentThread())
{
MOZ_ASSERT(
!NS_IsMainThread()); // This should be running on the worker thread
}
NS_IMETHOD
Run() override
{
MOZ_ASSERT(NS_IsMainThread()); // This method is supposed to run on the main
// thread!
// Declare javascript result events
RefPtr<SpeechEvent> event = new SpeechEvent(
mRecognition, SpeechRecognition::EVENT_RECOGNITIONSERVICE_FINAL_RESULT);
SpeechRecognitionResultList* resultList =
new SpeechRecognitionResultList(mRecognition);
SpeechRecognitionResult* result = new SpeechRecognitionResult(mRecognition);
if (0 < mRecognition->MaxAlternatives()) {
SpeechRecognitionAlternative* alternative =
new SpeechRecognitionAlternative(mRecognition);
alternative->mTranscript = mResult;
alternative->mConfidence = mConfidence;
result->mItems.AppendElement(alternative);
}
resultList->mItems.AppendElement(result);
event->mRecognitionResultList = resultList;
NS_DispatchToMainThread(event);
// If we don't destroy the thread when we're done with it, it will hang
// around forever... bad!
// But thread->Shutdown must be called from the main thread, not from the
// thread itself.
return mWorkerThread->Shutdown();
}
private:
nsString mResult;
float64 mConfidence;
WeakPtr<dom::SpeechRecognition> mRecognition;
nsCOMPtr<nsIThread> mWorkerThread;
};
class DecodeTask : public Runnable
{
public:
DecodeTask(WeakPtr<dom::SpeechRecognition> recogntion,
const nsTArray<int16_t>& audiovector, ps_decoder_t* ps)
: mRecognition(recogntion), mAudiovector(audiovector), mPs(ps)
{
}
NS_IMETHOD
Run() override
{
char const* hyp;
int rv;
int32 final;
int32 logprob;
float64 confidence;
nsAutoCString hypoValue;
rv = ps_start_utt(mPs);
rv = ps_process_raw(mPs, &mAudiovector[0], mAudiovector.Length(), FALSE,
FALSE);
rv = ps_end_utt(mPs);
confidence = 0;
if (rv >= 0) {
hyp = ps_get_hyp_final(mPs, &final);
if (hyp && final) {
logprob = ps_get_prob(mPs);
confidence = logmath_exp(ps_get_logmath(mPs), logprob);
hypoValue.Assign(hyp);
}
}
nsCOMPtr<nsIRunnable> resultrunnable =
new DecodeResultTask(NS_ConvertUTF8toUTF16(hypoValue), confidence, mRecognition);
return NS_DispatchToMainThread(resultrunnable);
}
private:
WeakPtr<dom::SpeechRecognition> mRecognition;
nsTArray<int16_t> mAudiovector;
ps_decoder_t* mPs;
};
NS_IMPL_ISUPPORTS(PocketSphinxSpeechRecognitionService,
nsISpeechRecognitionService, nsIObserver)
PocketSphinxSpeechRecognitionService::PocketSphinxSpeechRecognitionService()
{
mSpeexState = nullptr;
// get root folder
nsCOMPtr<nsIFile> tmpFile;
nsAutoString aStringAMPath; // am folder
nsAutoString aStringDictPath; // dict folder
NS_GetSpecialDirectory(NS_GRE_DIR, getter_AddRefs(tmpFile));
#if defined(XP_WIN) // for some reason, on windows NS_GRE_DIR is not bin root,
// but bin/browser
tmpFile->AppendRelativePath(NS_LITERAL_STRING(".."));
#endif
tmpFile->AppendRelativePath(NS_LITERAL_STRING("models"));
tmpFile->AppendRelativePath(NS_LITERAL_STRING("en-US"));
tmpFile->GetPath(aStringAMPath);
NS_GetSpecialDirectory(NS_GRE_DIR, getter_AddRefs(tmpFile));
#if defined(XP_WIN) // for some reason, on windows NS_GRE_DIR is not bin root,
// but bin/browser
tmpFile->AppendRelativePath(NS_LITERAL_STRING(".."));
#endif
tmpFile->AppendRelativePath(NS_LITERAL_STRING("models")); //
tmpFile->AppendRelativePath(NS_LITERAL_STRING("dict")); //
tmpFile->AppendRelativePath(NS_LITERAL_STRING("en-US.dic")); //
tmpFile->GetPath(aStringDictPath);
// FOR B2G PATHS HARDCODED (APPEND /DATA ON THE BEGINING, FOR DESKTOP, ONLY
// MODELS/ RELATIVE TO ROOT
mPSConfig = cmd_ln_init(nullptr, ps_args(), TRUE, "-bestpath", "yes", "-hmm",
ToNewUTF8String(aStringAMPath), // acoustic model
"-dict", ToNewUTF8String(aStringDictPath), nullptr);
if (mPSConfig == nullptr) {
ISDecoderCreated = false;
} else {
mPSHandle = ps_init(mPSConfig);
if (mPSHandle == nullptr) {
ISDecoderCreated = false;
} else {
ISDecoderCreated = true;
}
}
ISGrammarCompiled = false;
}
PocketSphinxSpeechRecognitionService::~PocketSphinxSpeechRecognitionService()
{
if (mPSConfig) {
free(mPSConfig);
}
if (mPSHandle) {
free(mPSHandle);
}
mSpeexState = nullptr;
}
// CALL START IN JS FALLS HERE
NS_IMETHODIMP
PocketSphinxSpeechRecognitionService::Initialize(
WeakPtr<SpeechRecognition> aSpeechRecognition)
{
if (!ISDecoderCreated || !ISGrammarCompiled) {
return NS_ERROR_NOT_INITIALIZED;
} else {
mAudioVector.Clear();
if (mSpeexState) {
mSpeexState = nullptr;
}
mRecognition = aSpeechRecognition;
nsCOMPtr<nsIObserverService> obs = services::GetObserverService();
obs->AddObserver(this, SPEECH_RECOGNITION_TEST_EVENT_REQUEST_TOPIC, false);
obs->AddObserver(this, SPEECH_RECOGNITION_TEST_END_TOPIC, false);
return NS_OK;
}
}
NS_IMETHODIMP
PocketSphinxSpeechRecognitionService::ProcessAudioSegment(
AudioSegment* aAudioSegment, int32_t aSampleRate)
{
if (!mSpeexState) {
mSpeexState = speex_resampler_init(1, aSampleRate, 16000,
SPEEX_RESAMPLER_QUALITY_MAX, nullptr);
}
aAudioSegment->ResampleChunks(mSpeexState, aSampleRate, 16000);
AudioSegment::ChunkIterator iterator(*aAudioSegment);
while (!iterator.IsEnded()) {
mozilla::AudioChunk& chunk = *(iterator);
MOZ_ASSERT(chunk.mBuffer);
const int16_t* buf = static_cast<const int16_t*>(chunk.mChannelData[0]);
for (int i = 0; i < iterator->mDuration; i++) {
mAudioVector.AppendElement((int16_t)buf[i]);
}
iterator.Next();
}
return NS_OK;
}
NS_IMETHODIMP
PocketSphinxSpeechRecognitionService::SoundEnd()
{
speex_resampler_destroy(mSpeexState);
mSpeexState = nullptr;
// To create a new thread, get the thread manager
nsCOMPtr<nsIThreadManager> tm = do_GetService(NS_THREADMANAGER_CONTRACTID);
nsCOMPtr<nsIThread> decodethread;
nsresult rv = tm->NewThread(0, 0, getter_AddRefs(decodethread));
if (NS_FAILED(rv)) {
// In case of failure, call back immediately with an empty string which
// indicates failure
return NS_OK;
}
nsCOMPtr<nsIRunnable> r =
new DecodeTask(mRecognition, mAudioVector, mPSHandle);
decodethread->Dispatch(r, nsIEventTarget::DISPATCH_NORMAL);
return NS_OK;
}
NS_IMETHODIMP
PocketSphinxSpeechRecognitionService::ValidateAndSetGrammarList(
SpeechGrammar* aSpeechGrammar,
nsISpeechGrammarCompilationCallback* aCallback)
{
if (!ISDecoderCreated) {
ISGrammarCompiled = false;
} else if (aSpeechGrammar) {
nsAutoString grammar;
ErrorResult rv;
aSpeechGrammar->GetSrc(grammar, rv);
int result = ps_set_jsgf_string(mPSHandle, "name",
NS_ConvertUTF16toUTF8(grammar).get());
if (result != 0) {
ISGrammarCompiled = false;
} else {
ps_set_search(mPSHandle, "name");
ISGrammarCompiled = true;
}
} else {
ISGrammarCompiled = false;
}
return ISGrammarCompiled ? NS_OK : NS_ERROR_NOT_INITIALIZED;
}
NS_IMETHODIMP
PocketSphinxSpeechRecognitionService::Abort()
{
return NS_OK;
}
NS_IMETHODIMP
PocketSphinxSpeechRecognitionService::Observe(nsISupports* aSubject,
const char* aTopic,
const char16_t* aData)
{
MOZ_ASSERT(MediaPrefs::WebSpeechFakeRecognitionService(),
"Got request to fake recognition service event, "
"but " TEST_PREFERENCE_FAKE_RECOGNITION_SERVICE " is not set");
if (!strcmp(aTopic, SPEECH_RECOGNITION_TEST_END_TOPIC)) {
nsCOMPtr<nsIObserverService> obs = services::GetObserverService();
obs->RemoveObserver(this, SPEECH_RECOGNITION_TEST_EVENT_REQUEST_TOPIC);
obs->RemoveObserver(this, SPEECH_RECOGNITION_TEST_END_TOPIC);
return NS_OK;
}
const nsDependentString eventName = nsDependentString(aData);
if (eventName.EqualsLiteral("EVENT_RECOGNITIONSERVICE_ERROR")) {
mRecognition->DispatchError(
SpeechRecognition::EVENT_RECOGNITIONSERVICE_ERROR,
SpeechRecognitionErrorCode::Network, // TODO different codes?
NS_LITERAL_STRING("RECOGNITIONSERVICE_ERROR test event"));
} else if (eventName.EqualsLiteral("EVENT_RECOGNITIONSERVICE_FINAL_RESULT")) {
RefPtr<SpeechEvent> event = new SpeechEvent(
mRecognition, SpeechRecognition::EVENT_RECOGNITIONSERVICE_FINAL_RESULT);
event->mRecognitionResultList = BuildMockResultList();
NS_DispatchToMainThread(event);
}
return NS_OK;
}
SpeechRecognitionResultList*
PocketSphinxSpeechRecognitionService::BuildMockResultList()
{
SpeechRecognitionResultList* resultList =
new SpeechRecognitionResultList(mRecognition);
SpeechRecognitionResult* result = new SpeechRecognitionResult(mRecognition);
if (0 < mRecognition->MaxAlternatives()) {
SpeechRecognitionAlternative* alternative =
new SpeechRecognitionAlternative(mRecognition);
alternative->mTranscript = NS_LITERAL_STRING("Mock final result");
alternative->mConfidence = 0.0f;
result->mItems.AppendElement(alternative);
}
resultList->mItems.AppendElement(result);
return resultList;
}
} // namespace mozilla

View file

@ -0,0 +1,85 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_PocketSphinxRecognitionService_h
#define mozilla_dom_PocketSphinxRecognitionService_h
#include "nsCOMPtr.h"
#include "nsTArray.h"
#include "nsIObserver.h"
#include "nsISpeechRecognitionService.h"
#include "speex/speex_resampler.h"
extern "C" {
#include <pocketsphinx/pocketsphinx.h>
#include <sphinxbase/sphinx_config.h>
}
#define NS_POCKETSPHINX_SPEECH_RECOGNITION_SERVICE_CID \
{ \
0x0ff5ce56, 0x5b09, 0x4db8, { \
0xad, 0xc6, 0x82, 0x66, 0xaf, 0x95, 0xf8, 0x64 \
} \
};
namespace mozilla {
/**
* Pocketsphix implementation of the nsISpeechRecognitionService interface
*/
class PocketSphinxSpeechRecognitionService : public nsISpeechRecognitionService,
public nsIObserver
{
public:
// Add XPCOM glue code
NS_DECL_ISUPPORTS
NS_DECL_NSISPEECHRECOGNITIONSERVICE
// Add nsIObserver code
NS_DECL_NSIOBSERVER
/**
* Default constructs a PocketSphinxSpeechRecognitionService loading default
* files
*/
PocketSphinxSpeechRecognitionService();
private:
/**
* Private destructor to prevent bypassing of reference counting
*/
virtual ~PocketSphinxSpeechRecognitionService();
/** The associated SpeechRecognition */
WeakPtr<dom::SpeechRecognition> mRecognition;
/**
* Builds a mock SpeechRecognitionResultList
*/
dom::SpeechRecognitionResultList* BuildMockResultList();
/** Speex state */
SpeexResamplerState* mSpeexState;
/** Pocksphix decoder */
ps_decoder_t* mPSHandle;
/** Sphinxbase parsed command line arguments */
cmd_ln_t* mPSConfig;
/** Flag to verify if decoder was created */
bool ISDecoderCreated;
/** Flag to verify if grammar was compiled */
bool ISGrammarCompiled;
/** Audio data */
nsTArray<int16_t> mAudioVector;
};
} // namespace mozilla
#endif

View file

@ -0,0 +1,81 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "SpeechGrammar.h"
#include "mozilla/dom/SpeechGrammarBinding.h"
namespace mozilla {
namespace dom {
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechGrammar, mParent)
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechGrammar)
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechGrammar)
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechGrammar)
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
NS_INTERFACE_MAP_ENTRY(nsISupports)
NS_INTERFACE_MAP_END
SpeechGrammar::SpeechGrammar(nsISupports* aParent)
: mParent(aParent)
{
}
SpeechGrammar::~SpeechGrammar()
{
}
already_AddRefed<SpeechGrammar>
SpeechGrammar::Constructor(const GlobalObject& aGlobal,
ErrorResult& aRv)
{
RefPtr<SpeechGrammar> speechGrammar =
new SpeechGrammar(aGlobal.GetAsSupports());
return speechGrammar.forget();
}
nsISupports*
SpeechGrammar::GetParentObject() const
{
return mParent;
}
JSObject*
SpeechGrammar::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
{
return SpeechGrammarBinding::Wrap(aCx, this, aGivenProto);
}
void
SpeechGrammar::GetSrc(nsString& aRetVal, ErrorResult& aRv) const
{
aRetVal = mSrc;
return;
}
void
SpeechGrammar::SetSrc(const nsAString& aArg, ErrorResult& aRv)
{
mSrc = aArg;
return;
}
float
SpeechGrammar::GetWeight(ErrorResult& aRv) const
{
aRv.Throw(NS_ERROR_NOT_IMPLEMENTED);
return 0;
}
void
SpeechGrammar::SetWeight(float aArg, ErrorResult& aRv)
{
aRv.Throw(NS_ERROR_NOT_IMPLEMENTED);
return;
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,59 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechGrammar_h
#define mozilla_dom_SpeechGrammar_h
#include "nsCOMPtr.h"
#include "nsCycleCollectionParticipant.h"
#include "nsString.h"
#include "nsWrapperCache.h"
#include "js/TypeDecls.h"
#include "mozilla/Attributes.h"
#include "mozilla/ErrorResult.h"
namespace mozilla {
namespace dom {
class GlobalObject;
class SpeechGrammar final : public nsISupports,
public nsWrapperCache
{
public:
explicit SpeechGrammar(nsISupports* aParent);
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechGrammar)
nsISupports* GetParentObject() const;
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
static already_AddRefed<SpeechGrammar>
Constructor(const GlobalObject& aGlobal, ErrorResult& aRv);
void GetSrc(nsString& aRetVal, ErrorResult& aRv) const;
void SetSrc(const nsAString& aArg, ErrorResult& aRv);
float GetWeight(ErrorResult& aRv) const;
void SetWeight(float aArg, ErrorResult& aRv);
private:
~SpeechGrammar();
nsCOMPtr<nsISupports> mParent;
nsString mSrc;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,103 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "SpeechGrammarList.h"
#include "mozilla/dom/SpeechGrammarListBinding.h"
#include "mozilla/ErrorResult.h"
#include "nsCOMPtr.h"
#include "nsXPCOMStrings.h"
#include "SpeechRecognition.h"
namespace mozilla {
namespace dom {
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechGrammarList, mParent, mItems)
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechGrammarList)
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechGrammarList)
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechGrammarList)
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
NS_INTERFACE_MAP_ENTRY(nsISupports)
NS_INTERFACE_MAP_END
SpeechGrammarList::SpeechGrammarList(nsISupports* aParent)
: mParent(aParent)
{
}
SpeechGrammarList::~SpeechGrammarList()
{
}
already_AddRefed<SpeechGrammarList>
SpeechGrammarList::Constructor(const GlobalObject& aGlobal,
ErrorResult& aRv)
{
RefPtr<SpeechGrammarList> speechGrammarList =
new SpeechGrammarList(aGlobal.GetAsSupports());
return speechGrammarList.forget();
}
JSObject*
SpeechGrammarList::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
{
return SpeechGrammarListBinding::Wrap(aCx, this, aGivenProto);
}
nsISupports*
SpeechGrammarList::GetParentObject() const
{
return mParent;
}
uint32_t
SpeechGrammarList::Length() const
{
return mItems.Length();
}
already_AddRefed<SpeechGrammar>
SpeechGrammarList::Item(uint32_t aIndex, ErrorResult& aRv)
{
RefPtr<SpeechGrammar> result = mItems.ElementAt(aIndex);
return result.forget();
}
void
SpeechGrammarList::AddFromURI(const nsAString& aSrc,
const Optional<float>& aWeight,
ErrorResult& aRv)
{
aRv.Throw(NS_ERROR_NOT_IMPLEMENTED);
return;
}
void
SpeechGrammarList::AddFromString(const nsAString& aString,
const Optional<float>& aWeight,
ErrorResult& aRv)
{
SpeechGrammar* speechGrammar = new SpeechGrammar(mParent);
speechGrammar->SetSrc(aString, aRv);
mItems.AppendElement(speechGrammar);
return;
}
already_AddRefed<SpeechGrammar>
SpeechGrammarList::IndexedGetter(uint32_t aIndex, bool& aPresent,
ErrorResult& aRv)
{
if (aIndex >= Length()) {
aPresent = false;
return nullptr;
}
ErrorResult rv;
aPresent = true;
return Item(aIndex, rv);
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,63 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechGrammarList_h
#define mozilla_dom_SpeechGrammarList_h
#include "mozilla/Attributes.h"
#include "nsCOMPtr.h"
#include "nsCycleCollectionParticipant.h"
#include "nsWrapperCache.h"
struct JSContext;
namespace mozilla {
class ErrorResult;
namespace dom {
class GlobalObject;
class SpeechGrammar;
template<typename> class Optional;
class SpeechGrammarList final : public nsISupports,
public nsWrapperCache
{
public:
explicit SpeechGrammarList(nsISupports* aParent);
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechGrammarList)
static already_AddRefed<SpeechGrammarList> Constructor(const GlobalObject& aGlobal, ErrorResult& aRv);
nsISupports* GetParentObject() const;
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
uint32_t Length() const;
already_AddRefed<SpeechGrammar> Item(uint32_t aIndex, ErrorResult& aRv);
void AddFromURI(const nsAString& aSrc, const Optional<float>& aWeight, ErrorResult& aRv);
void AddFromString(const nsAString& aString, const Optional<float>& aWeight, ErrorResult& aRv);
already_AddRefed<SpeechGrammar> IndexedGetter(uint32_t aIndex, bool& aPresent, ErrorResult& aRv);
private:
~SpeechGrammarList();
nsCOMPtr<nsISupports> mParent;
nsTArray<RefPtr<SpeechGrammar>> mItems;
};
} // namespace dom
} // namespace mozilla
#endif

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,296 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechRecognition_h
#define mozilla_dom_SpeechRecognition_h
#include "mozilla/Attributes.h"
#include "mozilla/DOMEventTargetHelper.h"
#include "nsCOMPtr.h"
#include "nsString.h"
#include "nsWrapperCache.h"
#include "nsTArray.h"
#include "js/TypeDecls.h"
#include "nsIDOMNavigatorUserMedia.h"
#include "nsITimer.h"
#include "MediaEngine.h"
#include "MediaStreamGraph.h"
#include "AudioSegment.h"
#include "mozilla/WeakPtr.h"
#include "SpeechGrammarList.h"
#include "SpeechRecognitionResultList.h"
#include "SpeechStreamListener.h"
#include "nsISpeechRecognitionService.h"
#include "endpointer.h"
#include "mozilla/dom/SpeechRecognitionError.h"
namespace mozilla {
namespace dom {
#define SPEECH_RECOGNITION_TEST_EVENT_REQUEST_TOPIC "SpeechRecognitionTest:RequestEvent"
#define SPEECH_RECOGNITION_TEST_END_TOPIC "SpeechRecognitionTest:End"
class GlobalObject;
class SpeechEvent;
LogModule* GetSpeechRecognitionLog();
#define SR_LOG(...) MOZ_LOG(GetSpeechRecognitionLog(), mozilla::LogLevel::Debug, (__VA_ARGS__))
class SpeechRecognition final : public DOMEventTargetHelper,
public nsIObserver,
public SupportsWeakPtr<SpeechRecognition>
{
public:
MOZ_DECLARE_WEAKREFERENCE_TYPENAME(SpeechRecognition)
explicit SpeechRecognition(nsPIDOMWindowInner* aOwnerWindow);
NS_DECL_ISUPPORTS_INHERITED
NS_DECL_CYCLE_COLLECTION_CLASS_INHERITED(SpeechRecognition, DOMEventTargetHelper)
NS_DECL_NSIOBSERVER
nsISupports* GetParentObject() const;
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
static bool IsAuthorized(JSContext* aCx, JSObject* aGlobal);
static already_AddRefed<SpeechRecognition>
Constructor(const GlobalObject& aGlobal, ErrorResult& aRv);
already_AddRefed<SpeechGrammarList> Grammars() const;
void SetGrammars(mozilla::dom::SpeechGrammarList& aArg);
void GetLang(nsString& aRetVal) const;
void SetLang(const nsAString& aArg);
bool GetContinuous(ErrorResult& aRv) const;
void SetContinuous(bool aArg, ErrorResult& aRv);
bool InterimResults() const;
void SetInterimResults(bool aArg);
uint32_t MaxAlternatives() const;
void SetMaxAlternatives(uint32_t aArg);
void GetServiceURI(nsString& aRetVal, ErrorResult& aRv) const;
void SetServiceURI(const nsAString& aArg, ErrorResult& aRv);
void Start(const Optional<NonNull<DOMMediaStream>>& aStream, ErrorResult& aRv);
void Stop();
void Abort();
IMPL_EVENT_HANDLER(audiostart)
IMPL_EVENT_HANDLER(soundstart)
IMPL_EVENT_HANDLER(speechstart)
IMPL_EVENT_HANDLER(speechend)
IMPL_EVENT_HANDLER(soundend)
IMPL_EVENT_HANDLER(audioend)
IMPL_EVENT_HANDLER(result)
IMPL_EVENT_HANDLER(nomatch)
IMPL_EVENT_HANDLER(error)
IMPL_EVENT_HANDLER(start)
IMPL_EVENT_HANDLER(end)
enum EventType {
EVENT_START,
EVENT_STOP,
EVENT_ABORT,
EVENT_AUDIO_DATA,
EVENT_AUDIO_ERROR,
EVENT_RECOGNITIONSERVICE_INTERMEDIATE_RESULT,
EVENT_RECOGNITIONSERVICE_FINAL_RESULT,
EVENT_RECOGNITIONSERVICE_ERROR,
EVENT_COUNT
};
void DispatchError(EventType aErrorType, SpeechRecognitionErrorCode aErrorCode, const nsAString& aMessage);
uint32_t FillSamplesBuffer(const int16_t* aSamples, uint32_t aSampleCount);
uint32_t SplitSamplesBuffer(const int16_t* aSamplesBuffer, uint32_t aSampleCount, nsTArray<RefPtr<SharedBuffer>>& aResult);
AudioSegment* CreateAudioSegment(nsTArray<RefPtr<SharedBuffer>>& aChunks);
void FeedAudioData(already_AddRefed<SharedBuffer> aSamples, uint32_t aDuration, MediaStreamListener* aProvider, TrackRate aTrackRate);
friend class SpeechEvent;
private:
virtual ~SpeechRecognition() {};
enum FSMState {
STATE_IDLE,
STATE_STARTING,
STATE_ESTIMATING,
STATE_WAITING_FOR_SPEECH,
STATE_RECOGNIZING,
STATE_WAITING_FOR_RESULT,
STATE_COUNT
};
void SetState(FSMState state);
bool StateBetween(FSMState begin, FSMState end);
bool SetRecognitionService(ErrorResult& aRv);
bool ValidateAndSetGrammarList(ErrorResult& aRv);
class GetUserMediaSuccessCallback : public nsIDOMGetUserMediaSuccessCallback
{
public:
NS_DECL_ISUPPORTS
NS_DECL_NSIDOMGETUSERMEDIASUCCESSCALLBACK
explicit GetUserMediaSuccessCallback(SpeechRecognition* aRecognition)
: mRecognition(aRecognition)
{}
private:
virtual ~GetUserMediaSuccessCallback() {}
RefPtr<SpeechRecognition> mRecognition;
};
class GetUserMediaErrorCallback : public nsIDOMGetUserMediaErrorCallback
{
public:
NS_DECL_ISUPPORTS
NS_DECL_NSIDOMGETUSERMEDIAERRORCALLBACK
explicit GetUserMediaErrorCallback(SpeechRecognition* aRecognition)
: mRecognition(aRecognition)
{}
private:
virtual ~GetUserMediaErrorCallback() {}
RefPtr<SpeechRecognition> mRecognition;
};
NS_IMETHOD StartRecording(DOMMediaStream* aDOMStream);
NS_IMETHOD StopRecording();
uint32_t ProcessAudioSegment(AudioSegment* aSegment, TrackRate aTrackRate);
void NotifyError(SpeechEvent* aEvent);
void ProcessEvent(SpeechEvent* aEvent);
void Transition(SpeechEvent* aEvent);
void Reset();
void ResetAndEnd();
void WaitForAudioData(SpeechEvent* aEvent);
void StartedAudioCapture(SpeechEvent* aEvent);
void StopRecordingAndRecognize(SpeechEvent* aEvent);
void WaitForEstimation(SpeechEvent* aEvent);
void DetectSpeech(SpeechEvent* aEvent);
void WaitForSpeechEnd(SpeechEvent* aEvent);
void NotifyFinalResult(SpeechEvent* aEvent);
void DoNothing(SpeechEvent* aEvent);
void AbortSilently(SpeechEvent* aEvent);
void AbortError(SpeechEvent* aEvent);
RefPtr<DOMMediaStream> mDOMStream;
RefPtr<SpeechStreamListener> mSpeechListener;
nsCOMPtr<nsISpeechRecognitionService> mRecognitionService;
FSMState mCurrentState;
Endpointer mEndpointer;
uint32_t mEstimationSamples;
uint32_t mAudioSamplesPerChunk;
// buffer holds one chunk of mAudioSamplesPerChunk
// samples before feeding it to mEndpointer
RefPtr<SharedBuffer> mAudioSamplesBuffer;
uint32_t mBufferedSamples;
nsCOMPtr<nsITimer> mSpeechDetectionTimer;
bool mAborted;
nsString mLang;
RefPtr<SpeechGrammarList> mSpeechGrammarList;
// WebSpeechAPI (http://bit.ly/1gIl7DC) states:
//
// 1. Default value MUST be false
// 2. If true, interim results SHOULD be returned
// 3. If false, interim results MUST NOT be returned
//
// Pocketsphinx does not return interm results; so, defaulting
// mInterimResults to false, then ignoring its subsequent value
// is a conforming implementation.
bool mInterimResults;
// WebSpeechAPI (http://bit.ly/1JAiqeo) states:
//
// 1. Default value is 1
// 2. Subsequent value is the "maximum number of SpeechRecognitionAlternatives per result"
//
// Pocketsphinx can only return at maximum a single SpeechRecognitionAlternative
// per SpeechRecognitionResult. So defaulting mMaxAlternatives to 1, for all non
// zero values ignoring mMaxAlternatives while for a 0 value returning no
// SpeechRecognitionAlternative per result is a conforming implementation.
uint32_t mMaxAlternatives;
void ProcessTestEventRequest(nsISupports* aSubject, const nsAString& aEventName);
const char* GetName(FSMState aId);
const char* GetName(SpeechEvent* aId);
};
class SpeechEvent : public Runnable
{
public:
SpeechEvent(SpeechRecognition* aRecognition, SpeechRecognition::EventType aType)
: mAudioSegment(0)
, mRecognitionResultList(nullptr)
, mError(nullptr)
, mRecognition(aRecognition)
, mType(aType)
, mTrackRate(0)
{
}
~SpeechEvent();
NS_IMETHOD Run() override;
AudioSegment* mAudioSegment;
RefPtr<SpeechRecognitionResultList> mRecognitionResultList; // TODO: make this a session being passed which also has index and stuff
RefPtr<SpeechRecognitionError> mError;
friend class SpeechRecognition;
private:
SpeechRecognition* mRecognition;
// for AUDIO_DATA events, keep a reference to the provider
// of the data (i.e., the SpeechStreamListener) to ensure it
// is kept alive (and keeps SpeechRecognition alive) until this
// event gets processed.
RefPtr<MediaStreamListener> mProvider;
SpeechRecognition::EventType mType;
TrackRate mTrackRate;
};
} // namespace dom
inline nsISupports*
ToSupports(dom::SpeechRecognition* aRec)
{
return ToSupports(static_cast<DOMEventTargetHelper*>(aRec));
}
} // namespace mozilla
#endif

View file

@ -0,0 +1,59 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "SpeechRecognitionAlternative.h"
#include "mozilla/dom/SpeechRecognitionAlternativeBinding.h"
#include "SpeechRecognition.h"
namespace mozilla {
namespace dom {
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechRecognitionAlternative, mParent)
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechRecognitionAlternative)
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechRecognitionAlternative)
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechRecognitionAlternative)
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
NS_INTERFACE_MAP_ENTRY(nsISupports)
NS_INTERFACE_MAP_END
SpeechRecognitionAlternative::SpeechRecognitionAlternative(SpeechRecognition* aParent)
: mConfidence(0)
, mParent(aParent)
{
}
SpeechRecognitionAlternative::~SpeechRecognitionAlternative()
{
}
JSObject*
SpeechRecognitionAlternative::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
{
return SpeechRecognitionAlternativeBinding::Wrap(aCx, this, aGivenProto);
}
nsISupports*
SpeechRecognitionAlternative::GetParentObject() const
{
return static_cast<DOMEventTargetHelper*>(mParent.get());
}
void
SpeechRecognitionAlternative::GetTranscript(nsString& aRetVal) const
{
aRetVal = mTranscript;
}
float
SpeechRecognitionAlternative::Confidence() const
{
return mConfidence;
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,50 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechRecognitionAlternative_h
#define mozilla_dom_SpeechRecognitionAlternative_h
#include "nsCycleCollectionParticipant.h"
#include "nsString.h"
#include "nsWrapperCache.h"
#include "js/TypeDecls.h"
#include "mozilla/Attributes.h"
namespace mozilla {
namespace dom {
class SpeechRecognition;
class SpeechRecognitionAlternative final : public nsISupports,
public nsWrapperCache
{
public:
explicit SpeechRecognitionAlternative(SpeechRecognition* aParent);
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechRecognitionAlternative)
nsISupports* GetParentObject() const;
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
void GetTranscript(nsString& aRetVal) const;
float Confidence() const;
nsString mTranscript;
float mConfidence;
private:
~SpeechRecognitionAlternative();
RefPtr<SpeechRecognition> mParent;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,76 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "SpeechRecognitionResult.h"
#include "mozilla/dom/SpeechRecognitionResultBinding.h"
#include "SpeechRecognition.h"
namespace mozilla {
namespace dom {
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechRecognitionResult, mParent)
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechRecognitionResult)
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechRecognitionResult)
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechRecognitionResult)
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
NS_INTERFACE_MAP_ENTRY(nsISupports)
NS_INTERFACE_MAP_END
SpeechRecognitionResult::SpeechRecognitionResult(SpeechRecognition* aParent)
: mParent(aParent)
{
}
SpeechRecognitionResult::~SpeechRecognitionResult()
{
}
JSObject*
SpeechRecognitionResult::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
{
return SpeechRecognitionResultBinding::Wrap(aCx, this, aGivenProto);
}
nsISupports*
SpeechRecognitionResult::GetParentObject() const
{
return static_cast<DOMEventTargetHelper*>(mParent.get());
}
already_AddRefed<SpeechRecognitionAlternative>
SpeechRecognitionResult::IndexedGetter(uint32_t aIndex, bool& aPresent)
{
if (aIndex >= Length()) {
aPresent = false;
return nullptr;
}
aPresent = true;
return Item(aIndex);
}
uint32_t
SpeechRecognitionResult::Length() const
{
return mItems.Length();
}
already_AddRefed<SpeechRecognitionAlternative>
SpeechRecognitionResult::Item(uint32_t aIndex)
{
RefPtr<SpeechRecognitionAlternative> alternative = mItems.ElementAt(aIndex);
return alternative.forget();
}
bool
SpeechRecognitionResult::IsFinal() const
{
return true; // TODO
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,55 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechRecognitionResult_h
#define mozilla_dom_SpeechRecognitionResult_h
#include "nsCOMPtr.h"
#include "nsCycleCollectionParticipant.h"
#include "nsWrapperCache.h"
#include "nsTArray.h"
#include "js/TypeDecls.h"
#include "mozilla/Attributes.h"
#include "SpeechRecognitionAlternative.h"
namespace mozilla {
namespace dom {
class SpeechRecognitionResult final : public nsISupports,
public nsWrapperCache
{
public:
explicit SpeechRecognitionResult(SpeechRecognition* aParent);
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechRecognitionResult)
nsISupports* GetParentObject() const;
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
uint32_t Length() const;
already_AddRefed<SpeechRecognitionAlternative> Item(uint32_t aIndex);
bool IsFinal() const;
already_AddRefed<SpeechRecognitionAlternative> IndexedGetter(uint32_t aIndex, bool& aPresent);
nsTArray<RefPtr<SpeechRecognitionAlternative>> mItems;
private:
~SpeechRecognitionResult();
RefPtr<SpeechRecognition> mParent;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,71 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "SpeechRecognitionResultList.h"
#include "mozilla/dom/SpeechRecognitionResultListBinding.h"
#include "SpeechRecognition.h"
namespace mozilla {
namespace dom {
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechRecognitionResultList, mParent, mItems)
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechRecognitionResultList)
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechRecognitionResultList)
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechRecognitionResultList)
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
NS_INTERFACE_MAP_ENTRY(nsISupports)
NS_INTERFACE_MAP_END
SpeechRecognitionResultList::SpeechRecognitionResultList(SpeechRecognition* aParent)
: mParent(aParent)
{
}
SpeechRecognitionResultList::~SpeechRecognitionResultList()
{
}
nsISupports*
SpeechRecognitionResultList::GetParentObject() const
{
return static_cast<DOMEventTargetHelper*>(mParent.get());
}
JSObject*
SpeechRecognitionResultList::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
{
return SpeechRecognitionResultListBinding::Wrap(aCx, this, aGivenProto);
}
already_AddRefed<SpeechRecognitionResult>
SpeechRecognitionResultList::IndexedGetter(uint32_t aIndex, bool& aPresent)
{
if (aIndex >= Length()) {
aPresent = false;
return nullptr;
}
aPresent = true;
return Item(aIndex);
}
uint32_t
SpeechRecognitionResultList::Length() const
{
return mItems.Length();
}
already_AddRefed<SpeechRecognitionResult>
SpeechRecognitionResultList::Item(uint32_t aIndex)
{
RefPtr<SpeechRecognitionResult> result = mItems.ElementAt(aIndex);
return result.forget();
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,53 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechRecognitionResultList_h
#define mozilla_dom_SpeechRecognitionResultList_h
#include "nsCycleCollectionParticipant.h"
#include "nsWrapperCache.h"
#include "nsTArray.h"
#include "js/TypeDecls.h"
#include "mozilla/Attributes.h"
#include "SpeechRecognitionResult.h"
namespace mozilla {
namespace dom {
class SpeechRecognition;
class SpeechRecognitionResultList final : public nsISupports,
public nsWrapperCache
{
public:
explicit SpeechRecognitionResultList(SpeechRecognition* aParent);
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechRecognitionResultList)
nsISupports* GetParentObject() const;
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
uint32_t Length() const;
already_AddRefed<SpeechRecognitionResult> Item(uint32_t aIndex);
already_AddRefed<SpeechRecognitionResult> IndexedGetter(uint32_t aIndex, bool& aPresent);
nsTArray<RefPtr<SpeechRecognitionResult>> mItems;
private:
~SpeechRecognitionResultList();
RefPtr<SpeechRecognition> mParent;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,94 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "SpeechStreamListener.h"
#include "SpeechRecognition.h"
#include "nsProxyRelease.h"
namespace mozilla {
namespace dom {
SpeechStreamListener::SpeechStreamListener(SpeechRecognition* aRecognition)
: mRecognition(aRecognition)
{
}
SpeechStreamListener::~SpeechStreamListener()
{
nsCOMPtr<nsIThread> mainThread;
NS_GetMainThread(getter_AddRefs(mainThread));
NS_ProxyRelease(mainThread, mRecognition.forget());
}
void
SpeechStreamListener::NotifyQueuedAudioData(MediaStreamGraph* aGraph, TrackID aID,
StreamTime aTrackOffset,
const AudioSegment& aQueuedMedia,
MediaStream* aInputStream,
TrackID aInputTrackID)
{
AudioSegment* audio = const_cast<AudioSegment*>(
static_cast<const AudioSegment*>(&aQueuedMedia));
AudioSegment::ChunkIterator iterator(*audio);
while (!iterator.IsEnded()) {
// Skip over-large chunks so we don't crash!
if (iterator->GetDuration() > INT_MAX) {
continue;
}
int duration = int(iterator->GetDuration());
if (iterator->IsNull()) {
nsTArray<int16_t> nullData;
PodZero(nullData.AppendElements(duration), duration);
ConvertAndDispatchAudioChunk(duration, iterator->mVolume,
nullData.Elements(), aGraph->GraphRate());
} else {
AudioSampleFormat format = iterator->mBufferFormat;
MOZ_ASSERT(format == AUDIO_FORMAT_S16 || format == AUDIO_FORMAT_FLOAT32);
if (format == AUDIO_FORMAT_S16) {
ConvertAndDispatchAudioChunk(duration,iterator->mVolume,
static_cast<const int16_t*>(iterator->mChannelData[0]),
aGraph->GraphRate());
} else if (format == AUDIO_FORMAT_FLOAT32) {
ConvertAndDispatchAudioChunk(duration,iterator->mVolume,
static_cast<const float*>(iterator->mChannelData[0]),
aGraph->GraphRate());
}
}
iterator.Next();
}
}
template<typename SampleFormatType> void
SpeechStreamListener::ConvertAndDispatchAudioChunk(int aDuration, float aVolume,
SampleFormatType* aData,
TrackRate aTrackRate)
{
RefPtr<SharedBuffer> samples(SharedBuffer::Create(aDuration *
1 * // channel
sizeof(int16_t)));
int16_t* to = static_cast<int16_t*>(samples->Data());
ConvertAudioSamplesWithScale(aData, to, aDuration, aVolume);
mRecognition->FeedAudioData(samples.forget(), aDuration, this, aTrackRate);
}
void
SpeechStreamListener::NotifyEvent(MediaStreamGraph* aGraph,
MediaStreamGraphEvent event)
{
// TODO dispatch SpeechEnd event so services can be informed
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,46 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechStreamListener_h
#define mozilla_dom_SpeechStreamListener_h
#include "MediaStreamGraph.h"
#include "MediaStreamListener.h"
#include "AudioSegment.h"
namespace mozilla {
class AudioSegment;
namespace dom {
class SpeechRecognition;
class SpeechStreamListener : public MediaStreamListener
{
public:
explicit SpeechStreamListener(SpeechRecognition* aRecognition);
~SpeechStreamListener();
void NotifyQueuedAudioData(MediaStreamGraph* aGraph, TrackID aID,
StreamTime aTrackOffset,
const AudioSegment& aQueuedMedia,
MediaStream* aInputStream,
TrackID aInputTrackID) override;
void NotifyEvent(MediaStreamGraph* aGraph,
MediaStreamGraphEvent event) override;
private:
template<typename SampleFormatType>
void ConvertAndDispatchAudioChunk(int aDuration, float aVolume, SampleFormatType* aData, TrackRate aTrackRate);
RefPtr<SpeechRecognition> mRecognition;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,193 @@
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
//
// Redistribution and use in source and binary forms, with or without
// modification, are permitted provided that the following conditions are
// met:
//
// * Redistributions of source code must retain the above copyright
// notice, this list of conditions and the following disclaimer.
// * Redistributions in binary form must reproduce the above
// copyright notice, this list of conditions and the following disclaimer
// in the documentation and/or other materials provided with the
// distribution.
// * Neither the name of Google Inc. nor the names of its
// contributors may be used to endorse or promote products derived from
// this software without specific prior written permission.
//
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "endpointer.h"
#include "AudioSegment.h"
namespace {
const int kFrameRate = 200; // 1 frame = 5ms of audio.
}
namespace mozilla {
Endpointer::Endpointer(int sample_rate)
: speech_input_possibly_complete_silence_length_us_(-1),
speech_input_complete_silence_length_us_(-1),
audio_frame_time_us_(0),
sample_rate_(sample_rate),
frame_size_(0) {
Reset();
frame_size_ = static_cast<int>(sample_rate / static_cast<float>(kFrameRate));
speech_input_minimum_length_us_ =
static_cast<int64_t>(1.7 * 1000000);
speech_input_complete_silence_length_us_ =
static_cast<int64_t>(0.5 * 1000000);
long_speech_input_complete_silence_length_us_ = -1;
long_speech_length_us_ = -1;
speech_input_possibly_complete_silence_length_us_ =
1 * 1000000;
// Set the default configuration for Push To Talk mode.
EnergyEndpointerParams ep_config;
ep_config.set_frame_period(1.0f / static_cast<float>(kFrameRate));
ep_config.set_frame_duration(1.0f / static_cast<float>(kFrameRate));
ep_config.set_endpoint_margin(0.2f);
ep_config.set_onset_window(0.15f);
ep_config.set_speech_on_window(0.4f);
ep_config.set_offset_window(0.15f);
ep_config.set_onset_detect_dur(0.09f);
ep_config.set_onset_confirm_dur(0.075f);
ep_config.set_on_maintain_dur(0.10f);
ep_config.set_offset_confirm_dur(0.12f);
ep_config.set_decision_threshold(1000.0f);
ep_config.set_min_decision_threshold(50.0f);
ep_config.set_fast_update_dur(0.2f);
ep_config.set_sample_rate(static_cast<float>(sample_rate));
ep_config.set_min_fundamental_frequency(57.143f);
ep_config.set_max_fundamental_frequency(400.0f);
ep_config.set_contamination_rejection_period(0.25f);
energy_endpointer_.Init(ep_config);
}
void Endpointer::Reset() {
old_ep_status_ = EP_PRE_SPEECH;
waiting_for_speech_possibly_complete_timeout_ = false;
waiting_for_speech_complete_timeout_ = false;
speech_previously_detected_ = false;
speech_input_complete_ = false;
audio_frame_time_us_ = 0; // Reset time for packets sent to endpointer.
speech_end_time_us_ = -1;
speech_start_time_us_ = -1;
}
void Endpointer::StartSession() {
Reset();
energy_endpointer_.StartSession();
}
void Endpointer::EndSession() {
energy_endpointer_.EndSession();
}
void Endpointer::SetEnvironmentEstimationMode() {
Reset();
energy_endpointer_.SetEnvironmentEstimationMode();
}
void Endpointer::SetUserInputMode() {
energy_endpointer_.SetUserInputMode();
}
EpStatus Endpointer::Status(int64_t *time) {
return energy_endpointer_.Status(time);
}
EpStatus Endpointer::ProcessAudio(const AudioChunk& raw_audio, float* rms_out) {
MOZ_ASSERT(raw_audio.mBufferFormat == AUDIO_FORMAT_S16, "Audio is not in 16 bit format");
const int16_t* audio_data = static_cast<const int16_t*>(raw_audio.mChannelData[0]);
const int num_samples = raw_audio.mDuration;
EpStatus ep_status = EP_PRE_SPEECH;
// Process the input data in blocks of frame_size_, dropping any incomplete
// frames at the end (which is ok since typically the caller will be recording
// audio in multiples of our frame size).
int sample_index = 0;
while (sample_index + frame_size_ <= num_samples) {
// Have the endpointer process the frame.
energy_endpointer_.ProcessAudioFrame(audio_frame_time_us_,
audio_data + sample_index,
frame_size_,
rms_out);
sample_index += frame_size_;
audio_frame_time_us_ += (frame_size_ * 1000000) /
sample_rate_;
// Get the status of the endpointer.
int64_t ep_time;
ep_status = energy_endpointer_.Status(&ep_time);
if (old_ep_status_ != ep_status)
fprintf(stderr, "Status changed old= %d, new= %d\n", old_ep_status_, ep_status);
// Handle state changes.
if ((EP_SPEECH_PRESENT == ep_status) &&
(EP_POSSIBLE_ONSET == old_ep_status_)) {
speech_end_time_us_ = -1;
waiting_for_speech_possibly_complete_timeout_ = false;
waiting_for_speech_complete_timeout_ = false;
// Trigger SpeechInputDidStart event on first detection.
if (false == speech_previously_detected_) {
speech_previously_detected_ = true;
speech_start_time_us_ = ep_time;
}
}
if ((EP_PRE_SPEECH == ep_status) &&
(EP_POSSIBLE_OFFSET == old_ep_status_)) {
speech_end_time_us_ = ep_time;
waiting_for_speech_possibly_complete_timeout_ = true;
waiting_for_speech_complete_timeout_ = true;
}
if (ep_time > speech_input_minimum_length_us_) {
// Speech possibly complete timeout.
if ((waiting_for_speech_possibly_complete_timeout_) &&
(ep_time - speech_end_time_us_ >
speech_input_possibly_complete_silence_length_us_)) {
waiting_for_speech_possibly_complete_timeout_ = false;
}
if (waiting_for_speech_complete_timeout_) {
// The length of the silence timeout period can be held constant, or it
// can be changed after a fixed amount of time from the beginning of
// speech.
bool has_stepped_silence =
(long_speech_length_us_ > 0) &&
(long_speech_input_complete_silence_length_us_ > 0);
int64_t requested_silence_length;
if (has_stepped_silence &&
(ep_time - speech_start_time_us_) > long_speech_length_us_) {
requested_silence_length =
long_speech_input_complete_silence_length_us_;
} else {
requested_silence_length =
speech_input_complete_silence_length_us_;
}
// Speech complete timeout.
if ((ep_time - speech_end_time_us_) > requested_silence_length) {
waiting_for_speech_complete_timeout_ = false;
speech_input_complete_ = true;
}
}
}
old_ep_status_ = ep_status;
}
return ep_status;
}
} // namespace mozilla

View file

@ -0,0 +1,180 @@
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
//
// Redistribution and use in source and binary forms, with or without
// modification, are permitted provided that the following conditions are
// met:
//
// * Redistributions of source code must retain the above copyright
// notice, this list of conditions and the following disclaimer.
// * Redistributions in binary form must reproduce the above
// copyright notice, this list of conditions and the following disclaimer
// in the documentation and/or other materials provided with the
// distribution.
// * Neither the name of Google Inc. nor the names of its
// contributors may be used to endorse or promote products derived from
// this software without specific prior written permission.
//
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef CONTENT_BROWSER_SPEECH_ENDPOINTER_ENDPOINTER_H_
#define CONTENT_BROWSER_SPEECH_ENDPOINTER_ENDPOINTER_H_
#include "energy_endpointer.h"
namespace mozilla {
struct AudioChunk;
// A simple interface to the underlying energy-endpointer implementation, this
// class lets callers provide audio as being recorded and let them poll to find
// when the user has stopped speaking.
//
// There are two events that may trigger the end of speech:
//
// speechInputPossiblyComplete event:
//
// Signals that silence/noise has been detected for a *short* amount of
// time after some speech has been detected. It can be used for low latency
// UI feedback. To disable it, set it to a large amount.
//
// speechInputComplete event:
//
// This event is intended to signal end of input and to stop recording.
// The amount of time to wait after speech is set by
// speech_input_complete_silence_length_ and optionally two other
// parameters (see below).
// This time can be held constant, or can change as more speech is detected.
// In the latter case, the time changes after a set amount of time from the
// *beginning* of speech. This is motivated by the expectation that there
// will be two distinct types of inputs: short search queries and longer
// dictation style input.
//
// Three parameters are used to define the piecewise constant timeout function.
// The timeout length is speech_input_complete_silence_length until
// long_speech_length, when it changes to
// long_speech_input_complete_silence_length.
class Endpointer {
public:
explicit Endpointer(int sample_rate);
// Start the endpointer. This should be called at the beginning of a session.
void StartSession();
// Stop the endpointer.
void EndSession();
// Start environment estimation. Audio will be used for environment estimation
// i.e. noise level estimation.
void SetEnvironmentEstimationMode();
// Start user input. This should be called when the user indicates start of
// input, e.g. by pressing a button.
void SetUserInputMode();
// Process a segment of audio, which may be more than one frame.
// The status of the last frame will be returned.
EpStatus ProcessAudio(const AudioChunk& raw_audio, float* rms_out);
// Get the status of the endpointer.
EpStatus Status(int64_t *time_us);
// Get the expected frame size for audio chunks. Audio chunks are expected
// to contain a number of samples that is a multiple of this number, and extra
// samples will be dropped.
int32_t FrameSize() const {
return frame_size_;
}
// Returns true if the endpointer detected reasonable audio levels above
// background noise which could be user speech, false if not.
bool DidStartReceivingSpeech() const {
return speech_previously_detected_;
}
bool IsEstimatingEnvironment() const {
return energy_endpointer_.estimating_environment();
}
void set_speech_input_complete_silence_length(int64_t time_us) {
speech_input_complete_silence_length_us_ = time_us;
}
void set_long_speech_input_complete_silence_length(int64_t time_us) {
long_speech_input_complete_silence_length_us_ = time_us;
}
void set_speech_input_possibly_complete_silence_length(int64_t time_us) {
speech_input_possibly_complete_silence_length_us_ = time_us;
}
void set_long_speech_length(int64_t time_us) {
long_speech_length_us_ = time_us;
}
bool speech_input_complete() const {
return speech_input_complete_;
}
// RMS background noise level in dB.
float NoiseLevelDb() const { return energy_endpointer_.GetNoiseLevelDb(); }
private:
// Reset internal states. Helper method common to initial input utterance
// and following input utternaces.
void Reset();
// Minimum allowable length of speech input.
int64_t speech_input_minimum_length_us_;
// The speechInputPossiblyComplete event signals that silence/noise has been
// detected for a *short* amount of time after some speech has been detected.
// This proporty specifies the time period.
int64_t speech_input_possibly_complete_silence_length_us_;
// The speechInputComplete event signals that silence/noise has been
// detected for a *long* amount of time after some speech has been detected.
// This property specifies the time period.
int64_t speech_input_complete_silence_length_us_;
// Same as above, this specifies the required silence period after speech
// detection. This period is used instead of
// speech_input_complete_silence_length_ when the utterance is longer than
// long_speech_length_. This parameter is optional.
int64_t long_speech_input_complete_silence_length_us_;
// The period of time after which the endpointer should consider
// long_speech_input_complete_silence_length_ as a valid silence period
// instead of speech_input_complete_silence_length_. This parameter is
// optional.
int64_t long_speech_length_us_;
// First speech onset time, used in determination of speech complete timeout.
int64_t speech_start_time_us_;
// Most recent end time, used in determination of speech complete timeout.
int64_t speech_end_time_us_;
int64_t audio_frame_time_us_;
EpStatus old_ep_status_;
bool waiting_for_speech_possibly_complete_timeout_;
bool waiting_for_speech_complete_timeout_;
bool speech_previously_detected_;
bool speech_input_complete_;
EnergyEndpointer energy_endpointer_;
int sample_rate_;
int32_t frame_size_;
};
} // namespace mozilla
#endif // CONTENT_BROWSER_SPEECH_ENDPOINTER_ENDPOINTER_H_

View file

@ -0,0 +1,393 @@
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
//
// Redistribution and use in source and binary forms, with or without
// modification, are permitted provided that the following conditions are
// met:
//
// * Redistributions of source code must retain the above copyright
// notice, this list of conditions and the following disclaimer.
// * Redistributions in binary form must reproduce the above
// copyright notice, this list of conditions and the following disclaimer
// in the documentation and/or other materials provided with the
// distribution.
// * Neither the name of Google Inc. nor the names of its
// contributors may be used to endorse or promote products derived from
// this software without specific prior written permission.
//
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "energy_endpointer.h"
#include <math.h>
namespace {
// Returns the RMS (quadratic mean) of the input signal.
float RMS(const int16_t* samples, int num_samples) {
int64_t ssq_int64_t = 0;
int64_t sum_int64_t = 0;
for (int i = 0; i < num_samples; ++i) {
sum_int64_t += samples[i];
ssq_int64_t += samples[i] * samples[i];
}
// now convert to floats.
double sum = static_cast<double>(sum_int64_t);
sum /= num_samples;
double ssq = static_cast<double>(ssq_int64_t);
return static_cast<float>(sqrt((ssq / num_samples) - (sum * sum)));
}
int64_t Secs2Usecs(float seconds) {
return static_cast<int64_t>(0.5 + (1.0e6 * seconds));
}
float GetDecibel(float value) {
if (value > 1.0e-100)
return 20 * log10(value);
return -2000.0;
}
} // namespace
namespace mozilla {
// Stores threshold-crossing histories for making decisions about the speech
// state.
class EnergyEndpointer::HistoryRing {
public:
HistoryRing() : insertion_index_(0) {}
// Resets the ring to |size| elements each with state |initial_state|
void SetRing(int size, bool initial_state);
// Inserts a new entry into the ring and drops the oldest entry.
void Insert(int64_t time_us, bool decision);
// Returns the time in microseconds of the most recently added entry.
int64_t EndTime() const;
// Returns the sum of all intervals during which 'decision' is true within
// the time in seconds specified by 'duration'. The returned interval is
// in seconds.
float RingSum(float duration_sec);
private:
struct DecisionPoint {
int64_t time_us;
bool decision;
};
std::vector<DecisionPoint> decision_points_;
int insertion_index_; // Index at which the next item gets added/inserted.
HistoryRing(const HistoryRing&);
void operator=(const HistoryRing&);
};
void EnergyEndpointer::HistoryRing::SetRing(int size, bool initial_state) {
insertion_index_ = 0;
decision_points_.clear();
DecisionPoint init = { -1, initial_state };
decision_points_.resize(size, init);
}
void EnergyEndpointer::HistoryRing::Insert(int64_t time_us, bool decision) {
decision_points_[insertion_index_].time_us = time_us;
decision_points_[insertion_index_].decision = decision;
insertion_index_ = (insertion_index_ + 1) % decision_points_.size();
}
int64_t EnergyEndpointer::HistoryRing::EndTime() const {
int ind = insertion_index_ - 1;
if (ind < 0)
ind = decision_points_.size() - 1;
return decision_points_[ind].time_us;
}
float EnergyEndpointer::HistoryRing::RingSum(float duration_sec) {
if (!decision_points_.size())
return 0.0;
int64_t sum_us = 0;
int ind = insertion_index_ - 1;
if (ind < 0)
ind = decision_points_.size() - 1;
int64_t end_us = decision_points_[ind].time_us;
bool is_on = decision_points_[ind].decision;
int64_t start_us = end_us - static_cast<int64_t>(0.5 + (1.0e6 * duration_sec));
if (start_us < 0)
start_us = 0;
size_t n_summed = 1; // n points ==> (n-1) intervals
while ((decision_points_[ind].time_us > start_us) &&
(n_summed < decision_points_.size())) {
--ind;
if (ind < 0)
ind = decision_points_.size() - 1;
if (is_on)
sum_us += end_us - decision_points_[ind].time_us;
is_on = decision_points_[ind].decision;
end_us = decision_points_[ind].time_us;
n_summed++;
}
return 1.0e-6f * sum_us; // Returns total time that was super threshold.
}
EnergyEndpointer::EnergyEndpointer()
: status_(EP_PRE_SPEECH),
offset_confirm_dur_sec_(0),
endpointer_time_us_(0),
fast_update_frames_(0),
frame_counter_(0),
max_window_dur_(4.0),
sample_rate_(0),
history_(new HistoryRing()),
decision_threshold_(0),
estimating_environment_(false),
noise_level_(0),
rms_adapt_(0),
start_lag_(0),
end_lag_(0),
user_input_start_time_us_(0) {
}
EnergyEndpointer::~EnergyEndpointer() {
}
int EnergyEndpointer::TimeToFrame(float time) const {
return static_cast<int32_t>(0.5 + (time / params_.frame_period()));
}
void EnergyEndpointer::Restart(bool reset_threshold) {
status_ = EP_PRE_SPEECH;
user_input_start_time_us_ = 0;
if (reset_threshold) {
decision_threshold_ = params_.decision_threshold();
rms_adapt_ = decision_threshold_;
noise_level_ = params_.decision_threshold() / 2.0f;
frame_counter_ = 0; // Used for rapid initial update of levels.
}
// Set up the memories to hold the history windows.
history_->SetRing(TimeToFrame(max_window_dur_), false);
// Flag that indicates that current input should be used for
// estimating the environment. The user has not yet started input
// by e.g. pressed the push-to-talk button. By default, this is
// false for backward compatibility.
estimating_environment_ = false;
}
void EnergyEndpointer::Init(const EnergyEndpointerParams& params) {
params_ = params;
// Find the longest history interval to be used, and make the ring
// large enough to accommodate that number of frames. NOTE: This
// depends upon ep_frame_period being set correctly in the factory
// that did this instantiation.
max_window_dur_ = params_.onset_window();
if (params_.speech_on_window() > max_window_dur_)
max_window_dur_ = params_.speech_on_window();
if (params_.offset_window() > max_window_dur_)
max_window_dur_ = params_.offset_window();
Restart(true);
offset_confirm_dur_sec_ = params_.offset_window() -
params_.offset_confirm_dur();
if (offset_confirm_dur_sec_ < 0.0)
offset_confirm_dur_sec_ = 0.0;
user_input_start_time_us_ = 0;
// Flag that indicates that current input should be used for
// estimating the environment. The user has not yet started input
// by e.g. pressed the push-to-talk button. By default, this is
// false for backward compatibility.
estimating_environment_ = false;
// The initial value of the noise and speech levels is inconsequential.
// The level of the first frame will overwrite these values.
noise_level_ = params_.decision_threshold() / 2.0f;
fast_update_frames_ =
static_cast<int64_t>(params_.fast_update_dur() / params_.frame_period());
frame_counter_ = 0; // Used for rapid initial update of levels.
sample_rate_ = params_.sample_rate();
start_lag_ = static_cast<int>(sample_rate_ /
params_.max_fundamental_frequency());
end_lag_ = static_cast<int>(sample_rate_ /
params_.min_fundamental_frequency());
}
void EnergyEndpointer::StartSession() {
Restart(true);
}
void EnergyEndpointer::EndSession() {
status_ = EP_POST_SPEECH;
}
void EnergyEndpointer::SetEnvironmentEstimationMode() {
Restart(true);
estimating_environment_ = true;
}
void EnergyEndpointer::SetUserInputMode() {
estimating_environment_ = false;
user_input_start_time_us_ = endpointer_time_us_;
}
void EnergyEndpointer::ProcessAudioFrame(int64_t time_us,
const int16_t* samples,
int num_samples,
float* rms_out) {
endpointer_time_us_ = time_us;
float rms = RMS(samples, num_samples);
// Check that this is user input audio vs. pre-input adaptation audio.
// Input audio starts when the user indicates start of input, by e.g.
// pressing push-to-talk. Audio recieved prior to that is used to update
// noise and speech level estimates.
if (!estimating_environment_) {
bool decision = false;
if ((endpointer_time_us_ - user_input_start_time_us_) <
Secs2Usecs(params_.contamination_rejection_period())) {
decision = false;
//PR_LOG(GetSpeechRecognitionLog(), PR_LOG_DEBUG, ("decision: forced to false, time: %d", endpointer_time_us_));
} else {
decision = (rms > decision_threshold_);
}
history_->Insert(endpointer_time_us_, decision);
switch (status_) {
case EP_PRE_SPEECH:
if (history_->RingSum(params_.onset_window()) >
params_.onset_detect_dur()) {
status_ = EP_POSSIBLE_ONSET;
}
break;
case EP_POSSIBLE_ONSET: {
float tsum = history_->RingSum(params_.onset_window());
if (tsum > params_.onset_confirm_dur()) {
status_ = EP_SPEECH_PRESENT;
} else { // If signal is not maintained, drop back to pre-speech.
if (tsum <= params_.onset_detect_dur())
status_ = EP_PRE_SPEECH;
}
break;
}
case EP_SPEECH_PRESENT: {
// To induce hysteresis in the state residency, we allow a
// smaller residency time in the on_ring, than was required to
// enter the SPEECH_PERSENT state.
float on_time = history_->RingSum(params_.speech_on_window());
if (on_time < params_.on_maintain_dur())
status_ = EP_POSSIBLE_OFFSET;
break;
}
case EP_POSSIBLE_OFFSET:
if (history_->RingSum(params_.offset_window()) <=
offset_confirm_dur_sec_) {
// Note that this offset time may be beyond the end
// of the input buffer in a real-time system. It will be up
// to the RecognizerSession to decide what to do.
status_ = EP_PRE_SPEECH; // Automatically reset for next utterance.
} else { // If speech picks up again we allow return to SPEECH_PRESENT.
if (history_->RingSum(params_.speech_on_window()) >=
params_.on_maintain_dur())
status_ = EP_SPEECH_PRESENT;
}
break;
default:
break;
}
// If this is a quiet, non-speech region, slowly adapt the detection
// threshold to be about 6dB above the average RMS.
if ((!decision) && (status_ == EP_PRE_SPEECH)) {
decision_threshold_ = (0.98f * decision_threshold_) + (0.02f * 2 * rms);
rms_adapt_ = decision_threshold_;
} else {
// If this is in a speech region, adapt the decision threshold to
// be about 10dB below the average RMS. If the noise level is high,
// the threshold is pushed up.
// Adaptation up to a higher level is 5 times faster than decay to
// a lower level.
if ((status_ == EP_SPEECH_PRESENT) && decision) {
if (rms_adapt_ > rms) {
rms_adapt_ = (0.99f * rms_adapt_) + (0.01f * rms);
} else {
rms_adapt_ = (0.95f * rms_adapt_) + (0.05f * rms);
}
float target_threshold = 0.3f * rms_adapt_ + noise_level_;
decision_threshold_ = (.90f * decision_threshold_) +
(0.10f * target_threshold);
}
}
// Set a floor
if (decision_threshold_ < params_.min_decision_threshold())
decision_threshold_ = params_.min_decision_threshold();
}
// Update speech and noise levels.
UpdateLevels(rms);
++frame_counter_;
if (rms_out)
*rms_out = GetDecibel(rms);
}
float EnergyEndpointer::GetNoiseLevelDb() const {
return GetDecibel(noise_level_);
}
void EnergyEndpointer::UpdateLevels(float rms) {
// Update quickly initially. We assume this is noise and that
// speech is 6dB above the noise.
if (frame_counter_ < fast_update_frames_) {
// Alpha increases from 0 to (k-1)/k where k is the number of time
// steps in the initial adaptation period.
float alpha = static_cast<float>(frame_counter_) /
static_cast<float>(fast_update_frames_);
noise_level_ = (alpha * noise_level_) + ((1 - alpha) * rms);
//PR_LOG(GetSpeechRecognitionLog(), PR_LOG_DEBUG, ("FAST UPDATE, frame_counter_ %d, fast_update_frames_ %d", frame_counter_, fast_update_frames_));
} else {
// Update Noise level. The noise level adapts quickly downward, but
// slowly upward. The noise_level_ parameter is not currently used
// for threshold adaptation. It is used for UI feedback.
if (noise_level_ < rms)
noise_level_ = (0.999f * noise_level_) + (0.001f * rms);
else
noise_level_ = (0.95f * noise_level_) + (0.05f * rms);
}
if (estimating_environment_ || (frame_counter_ < fast_update_frames_)) {
decision_threshold_ = noise_level_ * 2; // 6dB above noise level.
// Set a floor
if (decision_threshold_ < params_.min_decision_threshold())
decision_threshold_ = params_.min_decision_threshold();
}
}
EpStatus EnergyEndpointer::Status(int64_t* status_time) const {
*status_time = history_->EndTime();
return status_;
}
} // namespace mozilla

View file

@ -0,0 +1,180 @@
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
//
// Redistribution and use in source and binary forms, with or without
// modification, are permitted provided that the following conditions are
// met:
//
// * Redistributions of source code must retain the above copyright
// notice, this list of conditions and the following disclaimer.
// * Redistributions in binary form must reproduce the above
// copyright notice, this list of conditions and the following disclaimer
// in the documentation and/or other materials provided with the
// distribution.
// * Neither the name of Google Inc. nor the names of its
// contributors may be used to endorse or promote products derived from
// this software without specific prior written permission.
//
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
// The EnergyEndpointer class finds likely speech onset and offset points.
//
// The implementation described here is about the simplest possible.
// It is based on timings of threshold crossings for overall signal
// RMS. It is suitable for light weight applications.
//
// As written, the basic idea is that one specifies intervals that
// must be occupied by super- and sub-threshold energy levels, and
// defers decisions re onset and offset times until these
// specifications have been met. Three basic intervals are tested: an
// onset window, a speech-on window, and an offset window. We require
// super-threshold to exceed some mimimum total durations in the onset
// and speech-on windows before declaring the speech onset time, and
// we specify a required sub-threshold residency in the offset window
// before declaring speech offset. As the various residency requirements are
// met, the EnergyEndpointer instance assumes various states, and can return the
// ID of these states to the client (see EpStatus below).
//
// The levels of the speech and background noise are continuously updated. It is
// important that the background noise level be estimated initially for
// robustness in noisy conditions. The first frames are assumed to be background
// noise and a fast update rate is used for the noise level. The duration for
// fast update is controlled by the fast_update_dur_ paramter.
//
// If used in noisy conditions, the endpointer should be started and run in the
// EnvironmentEstimation mode, for at least 200ms, before switching to
// UserInputMode.
// Audio feedback contamination can appear in the input audio, if not cut
// out or handled by echo cancellation. Audio feedback can trigger a false
// accept. The false accepts can be ignored by setting
// ep_contamination_rejection_period.
#ifndef CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_H_
#define CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_H_
#include <vector>
#include "nsAutoPtr.h"
#include "energy_endpointer_params.h"
namespace mozilla {
// Endpointer status codes
enum EpStatus {
EP_PRE_SPEECH = 10,
EP_POSSIBLE_ONSET,
EP_SPEECH_PRESENT,
EP_POSSIBLE_OFFSET,
EP_POST_SPEECH,
};
class EnergyEndpointer {
public:
// The default construction MUST be followed by Init(), before any
// other use can be made of the instance.
EnergyEndpointer();
virtual ~EnergyEndpointer();
void Init(const EnergyEndpointerParams& params);
// Start the endpointer. This should be called at the beginning of a session.
void StartSession();
// Stop the endpointer.
void EndSession();
// Start environment estimation. Audio will be used for environment estimation
// i.e. noise level estimation.
void SetEnvironmentEstimationMode();
// Start user input. This should be called when the user indicates start of
// input, e.g. by pressing a button.
void SetUserInputMode();
// Computes the next input frame and modifies EnergyEndpointer status as
// appropriate based on the computation.
void ProcessAudioFrame(int64_t time_us,
const int16_t* samples, int num_samples,
float* rms_out);
// Returns the current state of the EnergyEndpointer and the time
// corresponding to the most recently computed frame.
EpStatus Status(int64_t* status_time_us) const;
bool estimating_environment() const {
return estimating_environment_;
}
// Returns estimated noise level in dB.
float GetNoiseLevelDb() const;
private:
class HistoryRing;
// Resets the endpointer internal state. If reset_threshold is true, the
// state will be reset completely, including adaptive thresholds and the
// removal of all history information.
void Restart(bool reset_threshold);
// Update internal speech and noise levels.
void UpdateLevels(float rms);
// Returns the number of frames (or frame number) corresponding to
// the 'time' (in seconds).
int TimeToFrame(float time) const;
EpStatus status_; // The current state of this instance.
float offset_confirm_dur_sec_; // max on time allowed to confirm POST_SPEECH
int64_t endpointer_time_us_; // Time of the most recently received audio frame.
int64_t fast_update_frames_; // Number of frames for initial level adaptation.
int64_t frame_counter_; // Number of frames seen. Used for initial adaptation.
float max_window_dur_; // Largest search window size (seconds)
float sample_rate_; // Sampling rate.
// Ring buffers to hold the speech activity history.
nsAutoPtr<HistoryRing> history_;
// Configuration parameters.
EnergyEndpointerParams params_;
// RMS which must be exceeded to conclude frame is speech.
float decision_threshold_;
// Flag to indicate that audio should be used to estimate environment, prior
// to receiving user input.
bool estimating_environment_;
// Estimate of the background noise level. Used externally for UI feedback.
float noise_level_;
// An adaptive threshold used to update decision_threshold_ when appropriate.
float rms_adapt_;
// Start lag corresponds to the highest fundamental frequency.
int start_lag_;
// End lag corresponds to the lowest fundamental frequency.
int end_lag_;
// Time when mode switched from environment estimation to user input. This
// is used to time forced rejection of audio feedback contamination.
int64_t user_input_start_time_us_;
// prevent copy constructor and assignment
EnergyEndpointer(const EnergyEndpointer&);
void operator=(const EnergyEndpointer&);
};
} // namespace mozilla
#endif // CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_H_

View file

@ -0,0 +1,77 @@
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
//
// Redistribution and use in source and binary forms, with or without
// modification, are permitted provided that the following conditions are
// met:
//
// * Redistributions of source code must retain the above copyright
// notice, this list of conditions and the following disclaimer.
// * Redistributions in binary form must reproduce the above
// copyright notice, this list of conditions and the following disclaimer
// in the documentation and/or other materials provided with the
// distribution.
// * Neither the name of Google Inc. nor the names of its
// contributors may be used to endorse or promote products derived from
// this software without specific prior written permission.
//
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "energy_endpointer_params.h"
namespace mozilla {
EnergyEndpointerParams::EnergyEndpointerParams() {
SetDefaults();
}
void EnergyEndpointerParams::SetDefaults() {
frame_period_ = 0.01f;
frame_duration_ = 0.01f;
endpoint_margin_ = 0.2f;
onset_window_ = 0.15f;
speech_on_window_ = 0.4f;
offset_window_ = 0.15f;
onset_detect_dur_ = 0.09f;
onset_confirm_dur_ = 0.075f;
on_maintain_dur_ = 0.10f;
offset_confirm_dur_ = 0.12f;
decision_threshold_ = 150.0f;
min_decision_threshold_ = 50.0f;
fast_update_dur_ = 0.2f;
sample_rate_ = 8000.0f;
min_fundamental_frequency_ = 57.143f;
max_fundamental_frequency_ = 400.0f;
contamination_rejection_period_ = 0.25f;
}
void EnergyEndpointerParams::operator=(const EnergyEndpointerParams& source) {
frame_period_ = source.frame_period();
frame_duration_ = source.frame_duration();
endpoint_margin_ = source.endpoint_margin();
onset_window_ = source.onset_window();
speech_on_window_ = source.speech_on_window();
offset_window_ = source.offset_window();
onset_detect_dur_ = source.onset_detect_dur();
onset_confirm_dur_ = source.onset_confirm_dur();
on_maintain_dur_ = source.on_maintain_dur();
offset_confirm_dur_ = source.offset_confirm_dur();
decision_threshold_ = source.decision_threshold();
min_decision_threshold_ = source.min_decision_threshold();
fast_update_dur_ = source.fast_update_dur();
sample_rate_ = source.sample_rate();
min_fundamental_frequency_ = source.min_fundamental_frequency();
max_fundamental_frequency_ = source.max_fundamental_frequency();
contamination_rejection_period_ = source.contamination_rejection_period();
}
} // namespace mozilla

View file

@ -0,0 +1,159 @@
// Copyright (c) 2013 The Chromium Authors. All rights reserved.
//
// Redistribution and use in source and binary forms, with or without
// modification, are permitted provided that the following conditions are
// met:
//
// * Redistributions of source code must retain the above copyright
// notice, this list of conditions and the following disclaimer.
// * Redistributions in binary form must reproduce the above
// copyright notice, this list of conditions and the following disclaimer
// in the documentation and/or other materials provided with the
// distribution.
// * Neither the name of Google Inc. nor the names of its
// contributors may be used to endorse or promote products derived from
// this software without specific prior written permission.
//
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_PARAMS_H_
#define CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_PARAMS_H_
namespace mozilla {
// Input parameters for the EnergyEndpointer class.
class EnergyEndpointerParams {
public:
EnergyEndpointerParams();
void SetDefaults();
void operator=(const EnergyEndpointerParams& source);
// Accessors and mutators
float frame_period() const { return frame_period_; }
void set_frame_period(float frame_period) {
frame_period_ = frame_period;
}
float frame_duration() const { return frame_duration_; }
void set_frame_duration(float frame_duration) {
frame_duration_ = frame_duration;
}
float endpoint_margin() const { return endpoint_margin_; }
void set_endpoint_margin(float endpoint_margin) {
endpoint_margin_ = endpoint_margin;
}
float onset_window() const { return onset_window_; }
void set_onset_window(float onset_window) { onset_window_ = onset_window; }
float speech_on_window() const { return speech_on_window_; }
void set_speech_on_window(float speech_on_window) {
speech_on_window_ = speech_on_window;
}
float offset_window() const { return offset_window_; }
void set_offset_window(float offset_window) {
offset_window_ = offset_window;
}
float onset_detect_dur() const { return onset_detect_dur_; }
void set_onset_detect_dur(float onset_detect_dur) {
onset_detect_dur_ = onset_detect_dur;
}
float onset_confirm_dur() const { return onset_confirm_dur_; }
void set_onset_confirm_dur(float onset_confirm_dur) {
onset_confirm_dur_ = onset_confirm_dur;
}
float on_maintain_dur() const { return on_maintain_dur_; }
void set_on_maintain_dur(float on_maintain_dur) {
on_maintain_dur_ = on_maintain_dur;
}
float offset_confirm_dur() const { return offset_confirm_dur_; }
void set_offset_confirm_dur(float offset_confirm_dur) {
offset_confirm_dur_ = offset_confirm_dur;
}
float decision_threshold() const { return decision_threshold_; }
void set_decision_threshold(float decision_threshold) {
decision_threshold_ = decision_threshold;
}
float min_decision_threshold() const { return min_decision_threshold_; }
void set_min_decision_threshold(float min_decision_threshold) {
min_decision_threshold_ = min_decision_threshold;
}
float fast_update_dur() const { return fast_update_dur_; }
void set_fast_update_dur(float fast_update_dur) {
fast_update_dur_ = fast_update_dur;
}
float sample_rate() const { return sample_rate_; }
void set_sample_rate(float sample_rate) { sample_rate_ = sample_rate; }
float min_fundamental_frequency() const { return min_fundamental_frequency_; }
void set_min_fundamental_frequency(float min_fundamental_frequency) {
min_fundamental_frequency_ = min_fundamental_frequency;
}
float max_fundamental_frequency() const { return max_fundamental_frequency_; }
void set_max_fundamental_frequency(float max_fundamental_frequency) {
max_fundamental_frequency_ = max_fundamental_frequency;
}
float contamination_rejection_period() const {
return contamination_rejection_period_;
}
void set_contamination_rejection_period(
float contamination_rejection_period) {
contamination_rejection_period_ = contamination_rejection_period;
}
private:
float frame_period_; // Frame period
float frame_duration_; // Window size
float onset_window_; // Interval scanned for onset activity
float speech_on_window_; // Inverval scanned for ongoing speech
float offset_window_; // Interval scanned for offset evidence
float offset_confirm_dur_; // Silence duration required to confirm offset
float decision_threshold_; // Initial rms detection threshold
float min_decision_threshold_; // Minimum rms detection threshold
float fast_update_dur_; // Period for initial estimation of levels.
float sample_rate_; // Expected sample rate.
// Time to add on either side of endpoint threshold crossings
float endpoint_margin_;
// Total dur within onset_window required to enter ONSET state
float onset_detect_dur_;
// Total on time within onset_window required to enter SPEECH_ON state
float onset_confirm_dur_;
// Minimum dur in SPEECH_ON state required to maintain ON state
float on_maintain_dur_;
// Minimum fundamental frequency for autocorrelation.
float min_fundamental_frequency_;
// Maximum fundamental frequency for autocorrelation.
float max_fundamental_frequency_;
// Period after start of user input that above threshold values are ignored.
// This is to reject audio feedback contamination.
float contamination_rejection_period_;
};
} // namespace mozilla
#endif // CONTENT_BROWSER_SPEECH_ENDPOINTER_ENERGY_ENDPOINTER_PARAMS_H_

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,34 @@
/* ====================================================================
* Copyright (c) 2015 Alpha Cephei Inc. All rights
* reserved.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
*
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
*
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in
* the documentation and/or other materials provided with the
* distribution.
*
* THIS SOFTWARE IS PROVIDED BY ALPHA CEPHEI INC. ``AS IS'' AND.
* ANY EXPRESSED OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO,.
* THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL ALPHA CEPHEI INC.
* NOR ITS EMPLOYEES BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
* SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT.
* LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,.
* DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY.
* THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT.
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE.
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*
* ====================================================================
*
*/
This directory contains generic US english acoustic model trained with
latest sphinxtrain.

View file

@ -0,0 +1,12 @@
-lowerf 130
-upperf 3700
-nfilt 20
-transform dct
-lifter 22
-feat 1s_c_d_dd
-svspec 0-12/13-25/26-38
-agc none
-cmn current
-varnorm no
-cmninit 30.09,0.49,5.51,7.43,-2.39,-1.64,-4.71,0.03,-5.11,0.20,4.84,6.14,0.35
-model ptm

File diff suppressed because it is too large Load diff

Binary file not shown.

View file

@ -0,0 +1,5 @@
<s> SIL
</s> SIL
<sil> SIL
[NOISE] +NSN+
[SPEECH] +SPN+

Binary file not shown.

Binary file not shown.

View file

@ -0,0 +1,88 @@
# vim: set filetype=python:
# This Source Code Form is subject to the terms of the Mozilla Public
# License, v. 2.0. If a copy of the MPL was not distributed with this
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
MOCHITEST_MANIFESTS += ['test/mochitest.ini']
XPIDL_MODULE = 'dom_webspeechrecognition'
XPIDL_SOURCES = [
'nsISpeechRecognitionService.idl'
]
EXPORTS.mozilla.dom += [
'SpeechGrammar.h',
'SpeechGrammarList.h',
'SpeechRecognition.h',
'SpeechRecognitionAlternative.h',
'SpeechRecognitionResult.h',
'SpeechRecognitionResultList.h',
'SpeechStreamListener.h',
]
if CONFIG['MOZ_WEBSPEECH_TEST_BACKEND']:
EXPORTS.mozilla.dom += [
'test/FakeSpeechRecognitionService.h',
]
if CONFIG['MOZ_WEBSPEECH_POCKETSPHINX']:
EXPORTS.mozilla.dom += [
'PocketSphinxSpeechRecognitionService.h',
]
UNIFIED_SOURCES += [
'endpointer.cc',
'energy_endpointer.cc',
'energy_endpointer_params.cc',
'SpeechGrammar.cpp',
'SpeechGrammarList.cpp',
'SpeechRecognition.cpp',
'SpeechRecognitionAlternative.cpp',
'SpeechRecognitionResult.cpp',
'SpeechRecognitionResultList.cpp',
'SpeechStreamListener.cpp',
]
if CONFIG['MOZ_WEBSPEECH_TEST_BACKEND']:
UNIFIED_SOURCES += [
'test/FakeSpeechRecognitionService.cpp',
]
if CONFIG['MOZ_WEBSPEECH_POCKETSPHINX']:
UNIFIED_SOURCES += [
'PocketSphinxSpeechRecognitionService.cpp',
]
LOCAL_INCLUDES += [
'/dom/base',
'/media/sphinxbase',
]
if CONFIG['MOZ_WEBSPEECH_POCKETSPHINX']:
LOCAL_INCLUDES += [
'/media/pocketsphinx',
]
if CONFIG['MOZ_WEBSPEECH_MODELS']:
FINAL_TARGET_FILES.models.dict += [
'models/dict/en-US.dic',
'models/dict/en-US.dic.dmp',
]
FINAL_TARGET_FILES.models['en-US'] += [
'models/en-US/feat.params',
'models/en-US/mdef',
'models/en-US/means',
'models/en-US/mixture_weights',
'models/en-US/noisedict',
'models/en-US/sendump',
'models/en-US/transition_matrices',
'models/en-US/variances',
]
include('/ipc/chromium/chromium-config.mozbuild')
FINAL_LIBRARY = 'xul'
if CONFIG['GNU_CXX']:
CXXFLAGS += ['-Wno-error=shadow']

View file

@ -0,0 +1,43 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 4 -*- */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsISupports.idl"
%{C++
#include "mozilla/WeakPtr.h"
namespace mozilla {
class AudioSegment;
namespace dom {
class SpeechRecognition;
class SpeechRecognitionResultList;
class SpeechGrammarList;
class SpeechGrammar;
}
}
%}
native SpeechRecognitionWeakPtr(mozilla::WeakPtr<mozilla::dom::SpeechRecognition>);
[ptr] native AudioSegmentPtr(mozilla::AudioSegment);
[ptr] native SpeechGrammarPtr(mozilla::dom::SpeechGrammar);
[ptr] native SpeechGrammarListPtr(mozilla::dom::SpeechGrammarList);
[uuid(6fcb6ee8-a6db-49ba-9f06-355d7ee18ea7)]
interface nsISpeechGrammarCompilationCallback : nsISupports {
void grammarCompilationEnd(in SpeechGrammarPtr grammarObject, in boolean success);
};
[uuid(8e97f287-f322-44e8-8888-8344fa408ef8)]
interface nsISpeechRecognitionService : nsISupports {
void initialize(in SpeechRecognitionWeakPtr aSpeechRecognition);
void processAudioSegment(in AudioSegmentPtr aAudioSegment, in long aSampleRate);
void validateAndSetGrammarList(in SpeechGrammarPtr aSpeechGrammar, in nsISpeechGrammarCompilationCallback aCallback);
void soundEnd();
void abort();
};
%{C++
#define NS_SPEECH_RECOGNITION_SERVICE_CONTRACTID_PREFIX "@mozilla.org/webspeech/service;1?name="
%}

View file

@ -0,0 +1,119 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsThreadUtils.h"
#include "FakeSpeechRecognitionService.h"
#include "MediaPrefs.h"
#include "SpeechRecognition.h"
#include "SpeechRecognitionAlternative.h"
#include "SpeechRecognitionResult.h"
#include "SpeechRecognitionResultList.h"
#include "nsIObserverService.h"
#include "mozilla/Services.h"
namespace mozilla {
using namespace dom;
NS_IMPL_ISUPPORTS(FakeSpeechRecognitionService, nsISpeechRecognitionService, nsIObserver)
FakeSpeechRecognitionService::FakeSpeechRecognitionService()
{
}
FakeSpeechRecognitionService::~FakeSpeechRecognitionService()
{
}
NS_IMETHODIMP
FakeSpeechRecognitionService::Initialize(WeakPtr<SpeechRecognition> aSpeechRecognition)
{
mRecognition = aSpeechRecognition;
nsCOMPtr<nsIObserverService> obs = services::GetObserverService();
obs->AddObserver(this, SPEECH_RECOGNITION_TEST_EVENT_REQUEST_TOPIC, false);
obs->AddObserver(this, SPEECH_RECOGNITION_TEST_END_TOPIC, false);
return NS_OK;
}
NS_IMETHODIMP
FakeSpeechRecognitionService::ProcessAudioSegment(AudioSegment* aAudioSegment, int32_t aSampleRate)
{
return NS_OK;
}
NS_IMETHODIMP
FakeSpeechRecognitionService::SoundEnd()
{
return NS_OK;
}
NS_IMETHODIMP
FakeSpeechRecognitionService::ValidateAndSetGrammarList(mozilla::dom::SpeechGrammar*, nsISpeechGrammarCompilationCallback*)
{
return NS_OK;
}
NS_IMETHODIMP
FakeSpeechRecognitionService::Abort()
{
return NS_OK;
}
NS_IMETHODIMP
FakeSpeechRecognitionService::Observe(nsISupports* aSubject, const char* aTopic, const char16_t* aData)
{
MOZ_ASSERT(MediaPrefs::WebSpeechFakeRecognitionService(),
"Got request to fake recognition service event, but "
TEST_PREFERENCE_FAKE_RECOGNITION_SERVICE " is not set");
if (!strcmp(aTopic, SPEECH_RECOGNITION_TEST_END_TOPIC)) {
nsCOMPtr<nsIObserverService> obs = services::GetObserverService();
obs->RemoveObserver(this, SPEECH_RECOGNITION_TEST_EVENT_REQUEST_TOPIC);
obs->RemoveObserver(this, SPEECH_RECOGNITION_TEST_END_TOPIC);
return NS_OK;
}
const nsDependentString eventName = nsDependentString(aData);
if (eventName.EqualsLiteral("EVENT_RECOGNITIONSERVICE_ERROR")) {
mRecognition->DispatchError(SpeechRecognition::EVENT_RECOGNITIONSERVICE_ERROR,
SpeechRecognitionErrorCode::Network, // TODO different codes?
NS_LITERAL_STRING("RECOGNITIONSERVICE_ERROR test event"));
} else if (eventName.EqualsLiteral("EVENT_RECOGNITIONSERVICE_FINAL_RESULT")) {
RefPtr<SpeechEvent> event =
new SpeechEvent(mRecognition,
SpeechRecognition::EVENT_RECOGNITIONSERVICE_FINAL_RESULT);
event->mRecognitionResultList = BuildMockResultList();
NS_DispatchToMainThread(event);
}
return NS_OK;
}
SpeechRecognitionResultList*
FakeSpeechRecognitionService::BuildMockResultList()
{
SpeechRecognitionResultList* resultList = new SpeechRecognitionResultList(mRecognition);
SpeechRecognitionResult* result = new SpeechRecognitionResult(mRecognition);
if (0 < mRecognition->MaxAlternatives()) {
SpeechRecognitionAlternative* alternative = new SpeechRecognitionAlternative(mRecognition);
alternative->mTranscript = NS_LITERAL_STRING("Mock final result");
alternative->mConfidence = 0.0f;
result->mItems.AppendElement(alternative);
}
resultList->mItems.AppendElement(result);
return resultList;
}
} // namespace mozilla

View file

@ -0,0 +1,38 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_FakeSpeechRecognitionService_h
#define mozilla_dom_FakeSpeechRecognitionService_h
#include "nsCOMPtr.h"
#include "nsIObserver.h"
#include "nsISpeechRecognitionService.h"
#define NS_FAKE_SPEECH_RECOGNITION_SERVICE_CID \
{0x48c345e7, 0x9929, 0x4f9a, {0xa5, 0x63, 0xf4, 0x78, 0x22, 0x2d, 0xab, 0xcd}};
namespace mozilla {
class FakeSpeechRecognitionService : public nsISpeechRecognitionService,
public nsIObserver
{
public:
NS_DECL_ISUPPORTS
NS_DECL_NSISPEECHRECOGNITIONSERVICE
NS_DECL_NSIOBSERVER
FakeSpeechRecognitionService();
private:
virtual ~FakeSpeechRecognitionService();
WeakPtr<dom::SpeechRecognition> mRecognition;
dom::SpeechRecognitionResultList* BuildMockResultList();
};
} // namespace mozilla
#endif

View file

@ -0,0 +1,181 @@
"use strict";
const DEFAULT_AUDIO_SAMPLE_FILE = "hello.ogg";
const SPEECH_RECOGNITION_TEST_REQUEST_EVENT_TOPIC = "SpeechRecognitionTest:RequestEvent";
const SPEECH_RECOGNITION_TEST_END_TOPIC = "SpeechRecognitionTest:End";
var errorCodes = {
NO_SPEECH : "no-speech",
ABORTED : "aborted",
AUDIO_CAPTURE : "audio-capture",
NETWORK : "network",
NOT_ALLOWED : "not-allowed",
SERVICE_NOT_ALLOWED : "service-not-allowed",
BAD_GRAMMAR : "bad-grammar",
LANGUAGE_NOT_SUPPORTED : "language-not-supported"
};
var Services = SpecialPowers.Cu.import("resource://gre/modules/Services.jsm").Services;
function EventManager(sr) {
var self = this;
var nEventsExpected = 0;
self.eventsReceived = [];
var allEvents = [
"audiostart",
"soundstart",
"speechstart",
"speechend",
"soundend",
"audioend",
"result",
"nomatch",
"error",
"start",
"end"
];
var eventDependencies = {
"speechend": "speechstart",
"soundend": "soundstart",
"audioend": "audiostart"
};
var isDone = false;
// set up grammar
var sgl = new SpeechGrammarList();
sgl.addFromString("#JSGF V1.0; grammar test; public <simple> = hello ;", 1);
sr.grammars = sgl;
// AUDIO_DATA events are asynchronous,
// so we queue events requested while they are being
// issued to make them seem synchronous
var isSendingAudioData = false;
var queuedEventRequests = [];
// register default handlers
for (var i = 0; i < allEvents.length; i++) {
(function (eventName) {
sr["on" + eventName] = function (evt) {
var message = "unexpected event: " + eventName;
if (eventName == "error") {
message += " -- " + evt.message;
}
ok(false, message);
if (self.doneFunc && !isDone) {
isDone = true;
self.doneFunc();
}
};
})(allEvents[i]);
}
self.expect = function EventManager_expect(eventName, cb) {
nEventsExpected++;
sr["on" + eventName] = function(evt) {
self.eventsReceived.push(eventName);
ok(true, "received event " + eventName);
var dep = eventDependencies[eventName];
if (dep) {
ok(self.eventsReceived.indexOf(dep) >= 0,
eventName + " must come after " + dep);
}
cb && cb(evt, sr);
if (self.doneFunc && !isDone &&
nEventsExpected === self.eventsReceived.length) {
isDone = true;
self.doneFunc();
}
}
}
self.start = function EventManager_start() {
isSendingAudioData = true;
var audioTag = document.createElement("audio");
audioTag.src = self.audioSampleFile;
var stream = audioTag.mozCaptureStreamUntilEnded();
audioTag.addEventListener("ended", function() {
info("Sample stream ended, requesting queued events");
isSendingAudioData = false;
while (queuedEventRequests.length) {
self.requestFSMEvent(queuedEventRequests.shift());
}
});
audioTag.play();
sr.start(stream);
}
self.requestFSMEvent = function EventManager_requestFSMEvent(eventName) {
if (isSendingAudioData) {
info("Queuing event " + eventName + " until we're done sending audio data");
queuedEventRequests.push(eventName);
return;
}
info("requesting " + eventName);
Services.obs.notifyObservers(null,
SPEECH_RECOGNITION_TEST_REQUEST_EVENT_TOPIC,
eventName);
}
self.requestTestEnd = function EventManager_requestTestEnd() {
Services.obs.notifyObservers(null, SPEECH_RECOGNITION_TEST_END_TOPIC, null);
}
}
function buildResultCallback(transcript) {
return (function(evt) {
is(evt.results[0][0].transcript, transcript, "expect correct transcript");
});
}
function buildErrorCallback(errcode) {
return (function(err) {
is(err.error, errcode, "expect correct error code");
});
}
function performTest(options) {
var prefs = options.prefs;
prefs.unshift(
["media.webspeech.recognition.enable", true],
["media.webspeech.test.enable", true]
);
SpecialPowers.pushPrefEnv({set: prefs}, function() {
var sr = new SpeechRecognition();
var em = new EventManager(sr);
for (var eventName in options.expectedEvents) {
var cb = options.expectedEvents[eventName];
em.expect(eventName, cb);
}
em.doneFunc = function() {
em.requestTestEnd();
if (options.doneFunc) {
options.doneFunc();
}
}
em.audioSampleFile = DEFAULT_AUDIO_SAMPLE_FILE;
if (options.audioSampleFile) {
em.audioSampleFile = options.audioSampleFile;
}
em.start();
for (var i = 0; i < options.eventsToRequest.length; i++) {
em.requestFSMEvent(options.eventsToRequest[i]);
}
});
}

Binary file not shown.

View file

@ -0,0 +1 @@
Cache-Control: no-store

View file

@ -0,0 +1,21 @@
[DEFAULT]
tags=msg
subsuite = media
support-files =
head.js
hello.ogg
hello.ogg^headers^
silence.ogg
silence.ogg^headers^
[test_abort.html]
skip-if = toolkit == 'android' # bug 1037287
[test_audio_capture_error.html]
[test_call_start_from_end_handler.html]
tags=capturestream
skip-if = (android_version == '18' && debug) # bug 967606
[test_nested_eventloop.html]
skip-if = toolkit == 'android'
[test_preference_enable.html]
[test_recognition_service_error.html]
[test_success_without_recognition_service.html]
[test_timeout.html]

Binary file not shown.

View file

@ -0,0 +1 @@
Cache-Control: no-store

View file

@ -0,0 +1,71 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 650295 -- Call abort from inside handlers</title>
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
<script type="application/javascript" src="head.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="text/javascript">
SimpleTest.waitForExplicitFinish();
// Abort inside event handlers, should't get a
// result after that
var nextEventIdx = 0;
var eventsToAbortOn = [
"start",
"audiostart",
"speechstart",
"speechend",
"audioend"
];
function doNextTest() {
var nextEvent = eventsToAbortOn[nextEventIdx];
var expectedEvents = {
"start": null,
"audiostart": null,
"audioend": null,
"end": null
};
if (nextEventIdx >= eventsToAbortOn.indexOf("speechstart")) {
expectedEvents["speechstart"] = null;
}
if (nextEventIdx >= eventsToAbortOn.indexOf("speechend")) {
expectedEvents["speechend"] = null;
}
info("Aborting on " + nextEvent);
expectedEvents[nextEvent] = function(evt, sr) {
sr.abort();
};
nextEventIdx++;
performTest({
eventsToRequest: [],
expectedEvents: expectedEvents,
doneFunc: (nextEventIdx < eventsToAbortOn.length) ? doNextTest : SimpleTest.finish,
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
});
}
doNextTest();
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,40 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 650295 -- Behavior on audio error</title>
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
<script type="application/javascript" src="head.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="text/javascript">
SimpleTest.waitForExplicitFinish();
performTest({
eventsToRequest: ['EVENT_AUDIO_ERROR'],
expectedEvents: {
'start': null,
'audiostart': null,
'speechstart': null,
'speechend': null,
'audioend': null,
'error': buildErrorCallback(errorCodes.AUDIO_CAPTURE),
'end': null
},
doneFunc: SimpleTest.finish,
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,100 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 650295 -- Restart recognition from end handler</title>
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
<script type="application/javascript" src="head.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="text/javascript">
SimpleTest.waitForExplicitFinish();
function createAudioStream() {
var audioTag = document.createElement("audio");
audioTag.src = DEFAULT_AUDIO_SAMPLE_FILE;
var stream = audioTag.mozCaptureStreamUntilEnded();
audioTag.play();
return stream;
}
var done = false;
function endHandler(evt, sr) {
if (done) {
SimpleTest.finish();
return;
}
try {
var stream = createAudioStream();
sr.start(stream); // shouldn't fail
} catch (err) {
ok(false, "Failed to start() from end() callback");
}
// calling start() may cause some callbacks to fire, but we're
// no longer interested in them, except for onend, which is where
// we'll conclude the test.
sr.onstart = null;
sr.onaudiostart = null;
sr.onspeechstart = null;
sr.onspeechend = null;
sr.onaudioend = null;
sr.onresult = null;
// FIXME(ggp) the state transition caused by start() is async,
// but abort() is sync (see bug 1055093). until we normalize
// state transitions, we need to setTimeout here to make sure
// abort() finds the speech recognition object in the correct
// state (namely, STATE_STARTING).
setTimeout(function() {
sr.abort();
done = true;
});
info("Successfully start() from end() callback");
}
function expectExceptionHandler(evt, sr) {
try {
sr.start(createAudioStream());
} catch (err) {
is(err.name, "InvalidStateError");
return;
}
ok(false, "Calling start() didn't raise InvalidStateError");
}
performTest({
eventsToRequest: [
'EVENT_RECOGNITIONSERVICE_FINAL_RESULT'
],
expectedEvents: {
'start': expectExceptionHandler,
'audiostart': expectExceptionHandler,
'speechstart': expectExceptionHandler,
'speechend': expectExceptionHandler,
'audioend': expectExceptionHandler,
'result': buildResultCallback("Mock final result"),
'end': endHandler,
},
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,81 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 650295 -- Spin the event loop from inside a callback</title>
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
<script type="application/javascript" src="head.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="text/javascript">
SimpleTest.waitForExplicitFinish();
/*
* window.showModalDialog() can be used to spin the event loop, causing
* queued SpeechEvents (such as those created by calls to start(), stop()
* or abort()) to be processed immediately.
* When this is done from inside DOM event handlers, it is possible to
* cause reentrancy in our C++ code, which we should be able to withstand.
*/
function abortAndSpinEventLoop(evt, sr) {
sr.abort();
SpecialPowers.spinEventLoop(window);
}
function doneFunc() {
// Trigger gc now and wait some time to make sure this test gets the blame
// for any assertions caused by showModalDialog
//
// NB - The assertions should be gone, but this looks too scary to touch
// during batch cleanup.
var count = 0, GC_COUNT = 4;
function triggerGCOrFinish() {
SpecialPowers.gc();
count++;
if (count == GC_COUNT) {
SimpleTest.finish();
}
}
for (var i = 0; i < GC_COUNT; i++) {
setTimeout(triggerGCOrFinish, 0);
}
}
/*
* We start by performing a normal start, then abort from the audiostart
* callback and force the EVENT_ABORT to be processed while still inside
* the event handler. This causes the recording to stop, which raises
* the audioend and (later on) end events.
* Then, we abort (once again spinning the event loop) from the audioend
* handler, attempting to cause a re-entry into the abort code. This second
* call should be ignored, and we get the end callback and finish.
*/
performTest({
eventsToRequest: [],
expectedEvents: {
"audiostart": abortAndSpinEventLoop,
"audioend": abortAndSpinEventLoop,
"end": null
},
doneFunc: doneFunc,
prefs: [["media.webspeech.test.fake_fsm_events", true],
["media.webspeech.test.fake_recognition_service", true]]
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,43 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 650295 -- No objects should be visible with preference disabled</title>
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="text/javascript">
SimpleTest.waitForExplicitFinish();
SpecialPowers.pushPrefEnv({
set: [["media.webspeech.recognition.enable", false]]
}, function() {
var objects = [
"SpeechRecognition",
"SpeechGrammar",
"SpeechRecognitionResult",
"SpeechRecognitionResultList",
"SpeechRecognitionAlternative"
];
for (var i = 0; i < objects.length; i++) {
is(window[objects[i]], undefined,
objects[i] + " should be undefined with pref off");
}
SimpleTest.finish();
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,43 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 650295 -- Behavior on recognition service error</title>
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
<script type="application/javascript" src="head.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="text/javascript">
SimpleTest.waitForExplicitFinish();
performTest({
eventsToRequest: [
'EVENT_RECOGNITIONSERVICE_ERROR'
],
expectedEvents: {
'start': null,
'audiostart': null,
'speechstart': null,
'speechend': null,
'audioend': null,
'error': buildErrorCallback(errorCodes.NETWORK),
'end': null
},
doneFunc: SimpleTest.finish,
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,43 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 650295 -- Success with fake recognition service</title>
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
<script type="application/javascript" src="head.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="text/javascript">
SimpleTest.waitForExplicitFinish();
performTest({
eventsToRequest: [
'EVENT_RECOGNITIONSERVICE_FINAL_RESULT'
],
expectedEvents: {
'start': null,
'audiostart': null,
'speechstart': null,
'speechend': null,
'audioend': null,
'result': buildResultCallback("Mock final result"),
'end': null
},
doneFunc:SimpleTest.finish,
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,40 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 650295 -- Timeout for user speech</title>
<script type="application/javascript" src="/tests/SimpleTest/SimpleTest.js"></script>
<link rel="stylesheet" type="text/css" href="/tests/SimpleTest/test.css"/>
<script type="application/javascript" src="head.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="text/javascript">
SimpleTest.waitForExplicitFinish();
performTest({
eventsToRequest: [],
expectedEvents: {
"start": null,
"audiostart": null,
"audioend": null,
"error": buildErrorCallback(errorCodes.NO_SPEECH),
"end": null
},
doneFunc: SimpleTest.finish,
audioSampleFile: "silence.ogg",
prefs: [["media.webspeech.test.fake_fsm_events", true], ["media.webspeech.test.fake_recognition_service", true]]
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,333 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsISupportsPrimitives.h"
#include "nsSpeechTask.h"
#include "mozilla/Logging.h"
#include "mozilla/dom/ContentChild.h"
#include "mozilla/dom/Element.h"
#include "mozilla/dom/SpeechSynthesisBinding.h"
#include "SpeechSynthesis.h"
#include "nsSynthVoiceRegistry.h"
#include "nsIDocument.h"
#undef LOG
mozilla::LogModule*
GetSpeechSynthLog()
{
static mozilla::LazyLogModule sLog("SpeechSynthesis");
return sLog;
}
#define LOG(type, msg) MOZ_LOG(GetSpeechSynthLog(), type, msg)
namespace mozilla {
namespace dom {
NS_IMPL_CYCLE_COLLECTION_CLASS(SpeechSynthesis)
NS_IMPL_CYCLE_COLLECTION_UNLINK_BEGIN_INHERITED(SpeechSynthesis, DOMEventTargetHelper)
NS_IMPL_CYCLE_COLLECTION_UNLINK(mCurrentTask)
NS_IMPL_CYCLE_COLLECTION_UNLINK(mSpeechQueue)
tmp->mVoiceCache.Clear();
NS_IMPL_CYCLE_COLLECTION_UNLINK_END
NS_IMPL_CYCLE_COLLECTION_TRAVERSE_BEGIN_INHERITED(SpeechSynthesis, DOMEventTargetHelper)
NS_IMPL_CYCLE_COLLECTION_TRAVERSE(mCurrentTask)
NS_IMPL_CYCLE_COLLECTION_TRAVERSE(mSpeechQueue)
for (auto iter = tmp->mVoiceCache.Iter(); !iter.Done(); iter.Next()) {
SpeechSynthesisVoice* voice = iter.UserData();
cb.NoteXPCOMChild(voice);
}
NS_IMPL_CYCLE_COLLECTION_TRAVERSE_END
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION_INHERITED(SpeechSynthesis)
NS_INTERFACE_MAP_ENTRY(nsIObserver)
NS_INTERFACE_MAP_ENTRY(nsISupportsWeakReference)
NS_INTERFACE_MAP_END_INHERITING(DOMEventTargetHelper)
NS_IMPL_ADDREF_INHERITED(SpeechSynthesis, DOMEventTargetHelper)
NS_IMPL_RELEASE_INHERITED(SpeechSynthesis, DOMEventTargetHelper)
SpeechSynthesis::SpeechSynthesis(nsPIDOMWindowInner* aParent)
: DOMEventTargetHelper(aParent)
, mHoldQueue(false)
, mInnerID(aParent->WindowID())
{
MOZ_ASSERT(aParent->IsInnerWindow());
MOZ_ASSERT(NS_IsMainThread());
nsCOMPtr<nsIObserverService> obs = mozilla::services::GetObserverService();
if (obs) {
obs->AddObserver(this, "inner-window-destroyed", true);
obs->AddObserver(this, "synth-voices-changed", true);
}
}
SpeechSynthesis::~SpeechSynthesis()
{
}
JSObject*
SpeechSynthesis::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
{
return SpeechSynthesisBinding::Wrap(aCx, this, aGivenProto);
}
bool
SpeechSynthesis::Pending() const
{
switch (mSpeechQueue.Length()) {
case 0:
return false;
case 1:
return mSpeechQueue.ElementAt(0)->GetState() == SpeechSynthesisUtterance::STATE_PENDING;
default:
return true;
}
}
bool
SpeechSynthesis::Speaking() const
{
if (!mSpeechQueue.IsEmpty() &&
mSpeechQueue.ElementAt(0)->GetState() == SpeechSynthesisUtterance::STATE_SPEAKING) {
return true;
}
// Returns global speaking state if global queue is enabled. Or false.
return nsSynthVoiceRegistry::GetInstance()->IsSpeaking();
}
bool
SpeechSynthesis::Paused() const
{
return mHoldQueue || (mCurrentTask && mCurrentTask->IsPrePaused()) ||
(!mSpeechQueue.IsEmpty() && mSpeechQueue.ElementAt(0)->IsPaused());
}
bool
SpeechSynthesis::HasEmptyQueue() const
{
return mSpeechQueue.Length() == 0;
}
bool SpeechSynthesis::HasVoices() const
{
uint32_t voiceCount = mVoiceCache.Count();
if (voiceCount == 0) {
nsresult rv = nsSynthVoiceRegistry::GetInstance()->GetVoiceCount(&voiceCount);
if(NS_WARN_IF(NS_FAILED(rv))) {
return false;
}
}
return voiceCount != 0;
}
void
SpeechSynthesis::Speak(SpeechSynthesisUtterance& aUtterance)
{
if (aUtterance.mState != SpeechSynthesisUtterance::STATE_NONE) {
// XXX: Should probably raise an error
return;
}
mSpeechQueue.AppendElement(&aUtterance);
aUtterance.mState = SpeechSynthesisUtterance::STATE_PENDING;
// If we only have one item in the queue, we aren't pre-paused, and
// we have voices available, speak it.
if (mSpeechQueue.Length() == 1 && !mCurrentTask && !mHoldQueue && HasVoices()) {
AdvanceQueue();
}
}
void
SpeechSynthesis::AdvanceQueue()
{
LOG(LogLevel::Debug,
("SpeechSynthesis::AdvanceQueue length=%d", mSpeechQueue.Length()));
if (mSpeechQueue.IsEmpty()) {
return;
}
RefPtr<SpeechSynthesisUtterance> utterance = mSpeechQueue.ElementAt(0);
nsAutoString docLang;
nsCOMPtr<nsPIDOMWindowInner> window = GetOwner();
nsIDocument* doc = window ? window->GetExtantDoc() : nullptr;
if (doc) {
Element* elm = doc->GetHtmlElement();
if (elm) {
elm->GetLang(docLang);
}
}
mCurrentTask =
nsSynthVoiceRegistry::GetInstance()->SpeakUtterance(*utterance, docLang);
if (mCurrentTask) {
mCurrentTask->SetSpeechSynthesis(this);
}
return;
}
void
SpeechSynthesis::Cancel()
{
if (!mSpeechQueue.IsEmpty() &&
mSpeechQueue.ElementAt(0)->GetState() == SpeechSynthesisUtterance::STATE_SPEAKING) {
// Remove all queued utterances except for current one, we will remove it
// in OnEnd
mSpeechQueue.RemoveElementsAt(1, mSpeechQueue.Length() - 1);
} else {
mSpeechQueue.Clear();
}
if (mCurrentTask) {
mCurrentTask->Cancel();
}
}
void
SpeechSynthesis::Pause()
{
if (Paused()) {
return;
}
if (mCurrentTask && !mSpeechQueue.IsEmpty() &&
mSpeechQueue.ElementAt(0)->GetState() != SpeechSynthesisUtterance::STATE_ENDED) {
mCurrentTask->Pause();
} else {
mHoldQueue = true;
}
}
void
SpeechSynthesis::Resume()
{
if (!Paused()) {
return;
}
if (mCurrentTask) {
mCurrentTask->Resume();
} else {
mHoldQueue = false;
AdvanceQueue();
}
}
void
SpeechSynthesis::OnEnd(const nsSpeechTask* aTask)
{
MOZ_ASSERT(mCurrentTask == aTask);
if (!mSpeechQueue.IsEmpty()) {
mSpeechQueue.RemoveElementAt(0);
}
mCurrentTask = nullptr;
AdvanceQueue();
}
void
SpeechSynthesis::GetVoices(nsTArray< RefPtr<SpeechSynthesisVoice> >& aResult)
{
aResult.Clear();
uint32_t voiceCount = 0;
nsresult rv = nsSynthVoiceRegistry::GetInstance()->GetVoiceCount(&voiceCount);
if(NS_WARN_IF(NS_FAILED(rv))) {
return;
}
nsISupports* voiceParent = NS_ISUPPORTS_CAST(nsIObserver*, this);
for (uint32_t i = 0; i < voiceCount; i++) {
nsAutoString uri;
rv = nsSynthVoiceRegistry::GetInstance()->GetVoice(i, uri);
if (NS_FAILED(rv)) {
NS_WARNING("Failed to retrieve voice from registry");
continue;
}
SpeechSynthesisVoice* voice = mVoiceCache.GetWeak(uri);
if (!voice) {
voice = new SpeechSynthesisVoice(voiceParent, uri);
}
aResult.AppendElement(voice);
}
mVoiceCache.Clear();
for (uint32_t i = 0; i < aResult.Length(); i++) {
SpeechSynthesisVoice* voice = aResult[i];
mVoiceCache.Put(voice->mUri, voice);
}
}
// For testing purposes, allows us to cancel the current task that is
// misbehaving, and flush the queue.
void
SpeechSynthesis::ForceEnd()
{
if (mCurrentTask) {
mCurrentTask->ForceEnd();
}
}
NS_IMETHODIMP
SpeechSynthesis::Observe(nsISupports* aSubject, const char* aTopic,
const char16_t* aData)
{
MOZ_ASSERT(NS_IsMainThread());
if (strcmp(aTopic, "inner-window-destroyed") == 0) {
nsCOMPtr<nsISupportsPRUint64> wrapper = do_QueryInterface(aSubject);
NS_ENSURE_TRUE(wrapper, NS_ERROR_FAILURE);
uint64_t innerID;
nsresult rv = wrapper->GetData(&innerID);
NS_ENSURE_SUCCESS(rv, rv);
if (innerID == mInnerID) {
Cancel();
nsCOMPtr<nsIObserverService> obs = mozilla::services::GetObserverService();
if (obs) {
obs->RemoveObserver(this, "inner-window-destroyed");
}
}
} else if (strcmp(aTopic, "synth-voices-changed") == 0) {
LOG(LogLevel::Debug, ("SpeechSynthesis::onvoiceschanged"));
DispatchTrustedEvent(NS_LITERAL_STRING("voiceschanged"));
// If we have a pending item, and voices become available, speak it.
if (!mCurrentTask && !mHoldQueue && HasVoices()) {
AdvanceQueue();
}
}
return NS_OK;
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,85 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechSynthesis_h
#define mozilla_dom_SpeechSynthesis_h
#include "nsCOMPtr.h"
#include "nsIObserver.h"
#include "nsRefPtrHashtable.h"
#include "nsString.h"
#include "nsWeakReference.h"
#include "nsWrapperCache.h"
#include "js/TypeDecls.h"
#include "SpeechSynthesisUtterance.h"
#include "SpeechSynthesisVoice.h"
class nsIDOMWindow;
namespace mozilla {
namespace dom {
class nsSpeechTask;
class SpeechSynthesis final : public DOMEventTargetHelper
, public nsIObserver
, public nsSupportsWeakReference
{
public:
explicit SpeechSynthesis(nsPIDOMWindowInner* aParent);
NS_DECL_ISUPPORTS_INHERITED
NS_DECL_CYCLE_COLLECTION_CLASS_INHERITED(SpeechSynthesis, DOMEventTargetHelper)
NS_DECL_NSIOBSERVER
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
bool Pending() const;
bool Speaking() const;
bool Paused() const;
bool HasEmptyQueue() const;
void Speak(SpeechSynthesisUtterance& aUtterance);
void Cancel();
void Pause();
void Resume();
void OnEnd(const nsSpeechTask* aTask);
void GetVoices(nsTArray< RefPtr<SpeechSynthesisVoice> >& aResult);
void ForceEnd();
IMPL_EVENT_HANDLER(voiceschanged)
private:
virtual ~SpeechSynthesis();
void AdvanceQueue();
bool HasVoices() const;
nsTArray<RefPtr<SpeechSynthesisUtterance> > mSpeechQueue;
RefPtr<nsSpeechTask> mCurrentTask;
nsRefPtrHashtable<nsStringHashKey, SpeechSynthesisVoice> mVoiceCache;
bool mHoldQueue;
uint64_t mInnerID;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,178 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsCOMPtr.h"
#include "nsCycleCollectionParticipant.h"
#include "nsGkAtoms.h"
#include "mozilla/dom/SpeechSynthesisEvent.h"
#include "mozilla/dom/SpeechSynthesisUtteranceBinding.h"
#include "SpeechSynthesisUtterance.h"
#include "SpeechSynthesisVoice.h"
#include <stdlib.h>
namespace mozilla {
namespace dom {
NS_IMPL_CYCLE_COLLECTION_INHERITED(SpeechSynthesisUtterance,
DOMEventTargetHelper, mVoice);
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION_INHERITED(SpeechSynthesisUtterance)
NS_INTERFACE_MAP_END_INHERITING(DOMEventTargetHelper)
NS_IMPL_ADDREF_INHERITED(SpeechSynthesisUtterance, DOMEventTargetHelper)
NS_IMPL_RELEASE_INHERITED(SpeechSynthesisUtterance, DOMEventTargetHelper)
SpeechSynthesisUtterance::SpeechSynthesisUtterance(nsPIDOMWindowInner* aOwnerWindow,
const nsAString& text)
: DOMEventTargetHelper(aOwnerWindow)
, mText(text)
, mVolume(1)
, mRate(1)
, mPitch(1)
, mState(STATE_NONE)
, mPaused(false)
{
}
SpeechSynthesisUtterance::~SpeechSynthesisUtterance() {}
JSObject*
SpeechSynthesisUtterance::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
{
return SpeechSynthesisUtteranceBinding::Wrap(aCx, this, aGivenProto);
}
nsISupports*
SpeechSynthesisUtterance::GetParentObject() const
{
return GetOwner();
}
already_AddRefed<SpeechSynthesisUtterance>
SpeechSynthesisUtterance::Constructor(GlobalObject& aGlobal,
ErrorResult& aRv)
{
return Constructor(aGlobal, EmptyString(), aRv);
}
already_AddRefed<SpeechSynthesisUtterance>
SpeechSynthesisUtterance::Constructor(GlobalObject& aGlobal,
const nsAString& aText,
ErrorResult& aRv)
{
nsCOMPtr<nsPIDOMWindowInner> win = do_QueryInterface(aGlobal.GetAsSupports());
if (!win) {
aRv.Throw(NS_ERROR_FAILURE);
}
MOZ_ASSERT(win->IsInnerWindow());
RefPtr<SpeechSynthesisUtterance> object =
new SpeechSynthesisUtterance(win, aText);
return object.forget();
}
void
SpeechSynthesisUtterance::GetText(nsString& aResult) const
{
aResult = mText;
}
void
SpeechSynthesisUtterance::SetText(const nsAString& aText)
{
mText = aText;
}
void
SpeechSynthesisUtterance::GetLang(nsString& aResult) const
{
aResult = mLang;
}
void
SpeechSynthesisUtterance::SetLang(const nsAString& aLang)
{
mLang = aLang;
}
SpeechSynthesisVoice*
SpeechSynthesisUtterance::GetVoice() const
{
return mVoice;
}
void
SpeechSynthesisUtterance::SetVoice(SpeechSynthesisVoice* aVoice)
{
mVoice = aVoice;
}
float
SpeechSynthesisUtterance::Volume() const
{
return mVolume;
}
void
SpeechSynthesisUtterance::SetVolume(float aVolume)
{
mVolume = std::max<float>(std::min<float>(aVolume, 1), 0);
}
float
SpeechSynthesisUtterance::Rate() const
{
return mRate;
}
void
SpeechSynthesisUtterance::SetRate(float aRate)
{
mRate = std::max<float>(std::min<float>(aRate, 10), 0.1f);
}
float
SpeechSynthesisUtterance::Pitch() const
{
return mPitch;
}
void
SpeechSynthesisUtterance::SetPitch(float aPitch)
{
mPitch = std::max<float>(std::min<float>(aPitch, 2), 0);
}
void
SpeechSynthesisUtterance::GetChosenVoiceURI(nsString& aResult) const
{
aResult = mChosenVoiceURI;
}
void
SpeechSynthesisUtterance::DispatchSpeechSynthesisEvent(const nsAString& aEventType,
uint32_t aCharIndex,
float aElapsedTime,
const nsAString& aName)
{
SpeechSynthesisEventInit init;
init.mBubbles = false;
init.mCancelable = false;
init.mUtterance = this;
init.mCharIndex = aCharIndex;
init.mElapsedTime = aElapsedTime;
init.mName = aName;
RefPtr<SpeechSynthesisEvent> event =
SpeechSynthesisEvent::Constructor(this, aEventType, init);
DispatchTrustedEvent(event);
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,124 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechSynthesisUtterance_h
#define mozilla_dom_SpeechSynthesisUtterance_h
#include "mozilla/DOMEventTargetHelper.h"
#include "nsCOMPtr.h"
#include "nsString.h"
#include "js/TypeDecls.h"
#include "nsSpeechTask.h"
namespace mozilla {
namespace dom {
class SpeechSynthesisVoice;
class SpeechSynthesis;
class nsSynthVoiceRegistry;
class SpeechSynthesisUtterance final : public DOMEventTargetHelper
{
friend class SpeechSynthesis;
friend class nsSpeechTask;
friend class nsSynthVoiceRegistry;
public:
SpeechSynthesisUtterance(nsPIDOMWindowInner* aOwnerWindow, const nsAString& aText);
NS_DECL_ISUPPORTS_INHERITED
NS_DECL_CYCLE_COLLECTION_CLASS_INHERITED(SpeechSynthesisUtterance,
DOMEventTargetHelper)
NS_REALLY_FORWARD_NSIDOMEVENTTARGET(DOMEventTargetHelper)
nsISupports* GetParentObject() const;
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
static
already_AddRefed<SpeechSynthesisUtterance> Constructor(GlobalObject& aGlobal,
ErrorResult& aRv);
static
already_AddRefed<SpeechSynthesisUtterance> Constructor(GlobalObject& aGlobal,
const nsAString& aText,
ErrorResult& aRv);
void GetText(nsString& aResult) const;
void SetText(const nsAString& aText);
void GetLang(nsString& aResult) const;
void SetLang(const nsAString& aLang);
SpeechSynthesisVoice* GetVoice() const;
void SetVoice(SpeechSynthesisVoice* aVoice);
float Volume() const;
void SetVolume(float aVolume);
float Rate() const;
void SetRate(float aRate);
float Pitch() const;
void SetPitch(float aPitch);
void GetChosenVoiceURI(nsString& aResult) const;
enum {
STATE_NONE,
STATE_PENDING,
STATE_SPEAKING,
STATE_ENDED
};
uint32_t GetState() { return mState; }
bool IsPaused() { return mPaused; }
IMPL_EVENT_HANDLER(start)
IMPL_EVENT_HANDLER(end)
IMPL_EVENT_HANDLER(error)
IMPL_EVENT_HANDLER(pause)
IMPL_EVENT_HANDLER(resume)
IMPL_EVENT_HANDLER(mark)
IMPL_EVENT_HANDLER(boundary)
private:
virtual ~SpeechSynthesisUtterance();
void DispatchSpeechSynthesisEvent(const nsAString& aEventType,
uint32_t aCharIndex,
float aElapsedTime, const nsAString& aName);
nsString mText;
nsString mLang;
float mVolume;
float mRate;
float mPitch;
nsString mChosenVoiceURI;
uint32_t mState;
bool mPaused;
RefPtr<SpeechSynthesisVoice> mVoice;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,94 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "SpeechSynthesis.h"
#include "nsSynthVoiceRegistry.h"
#include "mozilla/dom/SpeechSynthesisVoiceBinding.h"
namespace mozilla {
namespace dom {
NS_IMPL_CYCLE_COLLECTION_WRAPPERCACHE(SpeechSynthesisVoice, mParent)
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechSynthesisVoice)
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechSynthesisVoice)
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechSynthesisVoice)
NS_WRAPPERCACHE_INTERFACE_MAP_ENTRY
NS_INTERFACE_MAP_ENTRY(nsISupports)
NS_INTERFACE_MAP_END
SpeechSynthesisVoice::SpeechSynthesisVoice(nsISupports* aParent,
const nsAString& aUri)
: mParent(aParent)
, mUri(aUri)
{
}
SpeechSynthesisVoice::~SpeechSynthesisVoice()
{
}
JSObject*
SpeechSynthesisVoice::WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto)
{
return SpeechSynthesisVoiceBinding::Wrap(aCx, this, aGivenProto);
}
nsISupports*
SpeechSynthesisVoice::GetParentObject() const
{
return mParent;
}
void
SpeechSynthesisVoice::GetVoiceURI(nsString& aRetval) const
{
aRetval = mUri;
}
void
SpeechSynthesisVoice::GetName(nsString& aRetval) const
{
DebugOnly<nsresult> rv =
nsSynthVoiceRegistry::GetInstance()->GetVoiceName(mUri, aRetval);
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv),
"Failed to get SpeechSynthesisVoice.name");
}
void
SpeechSynthesisVoice::GetLang(nsString& aRetval) const
{
DebugOnly<nsresult> rv =
nsSynthVoiceRegistry::GetInstance()->GetVoiceLang(mUri, aRetval);
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv),
"Failed to get SpeechSynthesisVoice.lang");
}
bool
SpeechSynthesisVoice::LocalService() const
{
bool isLocal;
DebugOnly<nsresult> rv =
nsSynthVoiceRegistry::GetInstance()->IsLocalVoice(mUri, &isLocal);
NS_WARNING_ASSERTION(
NS_SUCCEEDED(rv), "Failed to get SpeechSynthesisVoice.localService");
return isLocal;
}
bool
SpeechSynthesisVoice::Default() const
{
bool isDefault;
DebugOnly<nsresult> rv =
nsSynthVoiceRegistry::GetInstance()->IsDefaultVoice(mUri, &isDefault);
NS_WARNING_ASSERTION(
NS_SUCCEEDED(rv), "Failed to get SpeechSynthesisVoice.default");
return isDefault;
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,60 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechSynthesisVoice_h
#define mozilla_dom_SpeechSynthesisVoice_h
#include "nsCOMPtr.h"
#include "nsString.h"
#include "nsWrapperCache.h"
#include "js/TypeDecls.h"
#include "nsISpeechService.h"
namespace mozilla {
namespace dom {
class nsSynthVoiceRegistry;
class SpeechSynthesis;
class SpeechSynthesisVoice final : public nsISupports,
public nsWrapperCache
{
friend class nsSynthVoiceRegistry;
friend class SpeechSynthesis;
public:
SpeechSynthesisVoice(nsISupports* aParent, const nsAString& aUri);
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
NS_DECL_CYCLE_COLLECTION_SCRIPT_HOLDER_CLASS(SpeechSynthesisVoice)
nsISupports* GetParentObject() const;
JSObject* WrapObject(JSContext* aCx, JS::Handle<JSObject*> aGivenProto) override;
void GetVoiceURI(nsString& aRetval) const;
void GetName(nsString& aRetval) const;
void GetLang(nsString& aRetval) const;
bool LocalService() const;
bool Default() const;
private:
virtual ~SpeechSynthesisVoice();
nsCOMPtr<nsISupports> mParent;
nsString mUri;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,57 @@
/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim: set ts=8 sts=2 et sw=2 tw=80: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "mozilla/ModuleUtils.h"
#include "nsIClassInfoImpl.h"
#include "OSXSpeechSynthesizerService.h"
using namespace mozilla::dom;
#define OSXSPEECHSYNTHESIZERSERVICE_CID \
{0x914e73b4, 0x6337, 0x4bef, {0x97, 0xf3, 0x4d, 0x06, 0x9e, 0x05, 0x3a, 0x12}}
#define OSXSPEECHSYNTHESIZERSERVICE_CONTRACTID "@mozilla.org/synthsystem;1"
// Defines OSXSpeechSynthesizerServiceConstructor
NS_GENERIC_FACTORY_SINGLETON_CONSTRUCTOR(OSXSpeechSynthesizerService,
OSXSpeechSynthesizerService::GetInstanceForService)
// Defines kOSXSERVICE_CID
NS_DEFINE_NAMED_CID(OSXSPEECHSYNTHESIZERSERVICE_CID);
static const mozilla::Module::CIDEntry kCIDs[] = {
{ &kOSXSPEECHSYNTHESIZERSERVICE_CID, true, nullptr, OSXSpeechSynthesizerServiceConstructor },
{ nullptr }
};
static const mozilla::Module::ContractIDEntry kContracts[] = {
{ OSXSPEECHSYNTHESIZERSERVICE_CONTRACTID, &kOSXSPEECHSYNTHESIZERSERVICE_CID },
{ nullptr }
};
static const mozilla::Module::CategoryEntry kCategories[] = {
{ "speech-synth-started", "OSX Speech Synth", OSXSPEECHSYNTHESIZERSERVICE_CONTRACTID },
{ nullptr }
};
static void
UnloadOSXSpeechSynthesizerModule()
{
OSXSpeechSynthesizerService::Shutdown();
}
static const mozilla::Module kModule = {
mozilla::Module::kVersion,
kCIDs,
kContracts,
kCategories,
nullptr,
nullptr,
UnloadOSXSpeechSynthesizerModule
};
NSMODULE_DEFN(osxsynth) = &kModule;

View file

@ -0,0 +1,44 @@
/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim: set ts=8 sts=2 et sw=2 tw=80: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_OsxSpeechSynthesizerService_h
#define mozilla_dom_OsxSpeechSynthesizerService_h
#include "nsISpeechService.h"
#include "nsIObserver.h"
#include "mozilla/StaticPtr.h"
namespace mozilla {
namespace dom {
class OSXSpeechSynthesizerService final : public nsISpeechService
, public nsIObserver
{
public:
NS_DECL_THREADSAFE_ISUPPORTS
NS_DECL_NSISPEECHSERVICE
NS_DECL_NSIOBSERVER
bool Init();
static OSXSpeechSynthesizerService* GetInstance();
static already_AddRefed<OSXSpeechSynthesizerService> GetInstanceForService();
static void Shutdown();
private:
OSXSpeechSynthesizerService();
virtual ~OSXSpeechSynthesizerService();
bool RegisterVoices();
bool mInitialized;
static mozilla::StaticRefPtr<OSXSpeechSynthesizerService> sSingleton;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,499 @@
/* -*- Mode: Objective-C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim: set ts=2 sw=2 et tw=80: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsISupports.h"
#include "nsServiceManagerUtils.h"
#include "nsObjCExceptions.h"
#include "nsCocoaUtils.h"
#include "nsThreadUtils.h"
#include "mozilla/dom/nsSynthVoiceRegistry.h"
#include "mozilla/dom/nsSpeechTask.h"
#include "mozilla/Preferences.h"
#include "mozilla/Assertions.h"
#include "OSXSpeechSynthesizerService.h"
#import <Cocoa/Cocoa.h>
// We can escape the default delimiters ("[[" and "]]") by temporarily
// changing the delimiters just before they appear, and changing them back
// just after.
#define DLIM_ESCAPE_START "[[dlim (( ))]]"
#define DLIM_ESCAPE_END "((dlim [[ ]]))"
using namespace mozilla;
class SpeechTaskCallback final : public nsISpeechTaskCallback
{
public:
SpeechTaskCallback(nsISpeechTask* aTask,
NSSpeechSynthesizer* aSynth,
const nsTArray<size_t>& aOffsets)
: mTask(aTask)
, mSpeechSynthesizer(aSynth)
, mOffsets(aOffsets)
{
mStartingTime = TimeStamp::Now();
}
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
NS_DECL_CYCLE_COLLECTION_CLASS_AMBIGUOUS(SpeechTaskCallback, nsISpeechTaskCallback)
NS_DECL_NSISPEECHTASKCALLBACK
void OnWillSpeakWord(uint32_t aIndex);
void OnError(uint32_t aIndex);
void OnDidFinishSpeaking();
private:
virtual ~SpeechTaskCallback()
{
[mSpeechSynthesizer release];
}
float GetTimeDurationFromStart();
nsCOMPtr<nsISpeechTask> mTask;
NSSpeechSynthesizer* mSpeechSynthesizer;
TimeStamp mStartingTime;
uint32_t mCurrentIndex;
nsTArray<size_t> mOffsets;
};
NS_IMPL_CYCLE_COLLECTION(SpeechTaskCallback, mTask);
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechTaskCallback)
NS_INTERFACE_MAP_ENTRY(nsISpeechTaskCallback)
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsISpeechTaskCallback)
NS_INTERFACE_MAP_END
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechTaskCallback)
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechTaskCallback)
NS_IMETHODIMP
SpeechTaskCallback::OnCancel()
{
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
[mSpeechSynthesizer stopSpeaking];
return NS_OK;
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
}
NS_IMETHODIMP
SpeechTaskCallback::OnPause()
{
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
[mSpeechSynthesizer pauseSpeakingAtBoundary:NSSpeechImmediateBoundary];
if (!mTask) {
// When calling pause() on child porcess, it may not receive end event
// from chrome process yet.
return NS_ERROR_FAILURE;
}
mTask->DispatchPause(GetTimeDurationFromStart(), mCurrentIndex);
return NS_OK;
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
}
NS_IMETHODIMP
SpeechTaskCallback::OnResume()
{
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
[mSpeechSynthesizer continueSpeaking];
if (!mTask) {
// When calling resume() on child porcess, it may not receive end event
// from chrome process yet.
return NS_ERROR_FAILURE;
}
mTask->DispatchResume(GetTimeDurationFromStart(), mCurrentIndex);
return NS_OK;
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
}
NS_IMETHODIMP
SpeechTaskCallback::OnVolumeChanged(float aVolume)
{
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
[mSpeechSynthesizer setObject:[NSNumber numberWithFloat:aVolume]
forProperty:NSSpeechVolumeProperty error:nil];
return NS_OK;
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
}
float
SpeechTaskCallback::GetTimeDurationFromStart()
{
TimeDuration duration = TimeStamp::Now() - mStartingTime;
return duration.ToMilliseconds();
}
void
SpeechTaskCallback::OnWillSpeakWord(uint32_t aIndex)
{
mCurrentIndex = aIndex < mOffsets.Length() ? mOffsets[aIndex] : mCurrentIndex;
if (!mTask) {
return;
}
mTask->DispatchBoundary(NS_LITERAL_STRING("word"),
GetTimeDurationFromStart(), mCurrentIndex);
}
void
SpeechTaskCallback::OnError(uint32_t aIndex)
{
if (!mTask) {
return;
}
mTask->DispatchError(GetTimeDurationFromStart(), aIndex);
}
void
SpeechTaskCallback::OnDidFinishSpeaking()
{
mTask->DispatchEnd(GetTimeDurationFromStart(), mCurrentIndex);
// no longer needed
[mSpeechSynthesizer setDelegate:nil];
mTask = nullptr;
}
@interface SpeechDelegate : NSObject<NSSpeechSynthesizerDelegate>
{
@private
SpeechTaskCallback* mCallback;
}
- (id)initWithCallback:(SpeechTaskCallback*)aCallback;
@end
@implementation SpeechDelegate
- (id)initWithCallback:(SpeechTaskCallback*)aCallback
{
[super init];
mCallback = aCallback;
return self;
}
- (void)speechSynthesizer:(NSSpeechSynthesizer *)aSender
willSpeakWord:(NSRange)aRange ofString:(NSString*)aString
{
mCallback->OnWillSpeakWord(aRange.location);
}
- (void)speechSynthesizer:(NSSpeechSynthesizer *)aSender
didFinishSpeaking:(BOOL)aFinishedSpeaking
{
mCallback->OnDidFinishSpeaking();
}
- (void)speechSynthesizer:(NSSpeechSynthesizer*)aSender
didEncounterErrorAtIndex:(NSUInteger)aCharacterIndex
ofString:(NSString*)aString
message:(NSString*)aMessage
{
mCallback->OnError(aCharacterIndex);
}
@end
namespace mozilla {
namespace dom {
struct OSXVoice
{
OSXVoice() : mIsDefault(false)
{
}
nsString mUri;
nsString mName;
nsString mLocale;
bool mIsDefault;
};
class RegisterVoicesRunnable final : public Runnable
{
public:
RegisterVoicesRunnable(OSXSpeechSynthesizerService* aSpeechService,
nsTArray<OSXVoice>& aList)
: mSpeechService(aSpeechService)
, mVoices(aList)
{
}
NS_IMETHOD Run() override;
private:
~RegisterVoicesRunnable()
{
}
// This runnable always use sync mode. It is unnecesarry to reference object
OSXSpeechSynthesizerService* mSpeechService;
nsTArray<OSXVoice>& mVoices;
};
NS_IMETHODIMP
RegisterVoicesRunnable::Run()
{
nsresult rv;
nsCOMPtr<nsISynthVoiceRegistry> registry =
do_GetService(NS_SYNTHVOICEREGISTRY_CONTRACTID, &rv);
if (!registry) {
return rv;
}
for (OSXVoice voice : mVoices) {
rv = registry->AddVoice(mSpeechService, voice.mUri, voice.mName, voice.mLocale, true, false);
if (NS_WARN_IF(NS_FAILED(rv))) {
continue;
}
if (voice.mIsDefault) {
registry->SetDefaultVoice(voice.mUri, true);
}
}
registry->NotifyVoicesChanged();
return NS_OK;
}
class EnumVoicesRunnable final : public Runnable
{
public:
explicit EnumVoicesRunnable(OSXSpeechSynthesizerService* aSpeechService)
: mSpeechService(aSpeechService)
{
}
NS_IMETHOD Run() override;
private:
~EnumVoicesRunnable()
{
}
RefPtr<OSXSpeechSynthesizerService> mSpeechService;
};
NS_IMETHODIMP
EnumVoicesRunnable::Run()
{
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
AutoTArray<OSXVoice, 64> list;
NSArray* voices = [NSSpeechSynthesizer availableVoices];
NSString* defaultVoice = [NSSpeechSynthesizer defaultVoice];
for (NSString* voice in voices) {
OSXVoice item;
NSDictionary* attr = [NSSpeechSynthesizer attributesForVoice:voice];
nsAutoString identifier;
nsCocoaUtils::GetStringForNSString([attr objectForKey:NSVoiceIdentifier],
identifier);
nsCocoaUtils::GetStringForNSString([attr objectForKey:NSVoiceName], item.mName);
nsCocoaUtils::GetStringForNSString(
[attr objectForKey:NSVoiceLocaleIdentifier], item.mLocale);
item.mLocale.ReplaceChar('_', '-');
item.mUri.AssignLiteral("urn:moz-tts:osx:");
item.mUri.Append(identifier);
if ([voice isEqualToString:defaultVoice]) {
item.mIsDefault = true;
}
list.AppendElement(item);
}
RefPtr<RegisterVoicesRunnable> runnable = new RegisterVoicesRunnable(mSpeechService, list);
NS_DispatchToMainThread(runnable, NS_DISPATCH_SYNC);
return NS_OK;
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
}
StaticRefPtr<OSXSpeechSynthesizerService> OSXSpeechSynthesizerService::sSingleton;
NS_INTERFACE_MAP_BEGIN(OSXSpeechSynthesizerService)
NS_INTERFACE_MAP_ENTRY(nsISpeechService)
NS_INTERFACE_MAP_ENTRY(nsIObserver)
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsISpeechService)
NS_INTERFACE_MAP_END
NS_IMPL_ADDREF(OSXSpeechSynthesizerService)
NS_IMPL_RELEASE(OSXSpeechSynthesizerService)
OSXSpeechSynthesizerService::OSXSpeechSynthesizerService()
: mInitialized(false)
{
}
OSXSpeechSynthesizerService::~OSXSpeechSynthesizerService()
{
}
bool
OSXSpeechSynthesizerService::Init()
{
if (Preferences::GetBool("media.webspeech.synth.test") ||
!Preferences::GetBool("media.webspeech.synth.enabled")) {
// When test is enabled, we shouldn't add OS backend (Bug 1160844)
return false;
}
nsCOMPtr<nsIThread> thread;
if (NS_FAILED(NS_NewNamedThread("SpeechWorker", getter_AddRefs(thread)))) {
return false;
}
// Get all the voices and register in the SynthVoiceRegistry
nsCOMPtr<nsIRunnable> runnable = new EnumVoicesRunnable(this);
thread->Dispatch(runnable, NS_DISPATCH_NORMAL);
mInitialized = true;
return true;
}
NS_IMETHODIMP
OSXSpeechSynthesizerService::Speak(const nsAString& aText,
const nsAString& aUri,
float aVolume,
float aRate,
float aPitch,
nsISpeechTask* aTask)
{
NS_OBJC_BEGIN_TRY_ABORT_BLOCK_NSRESULT;
MOZ_ASSERT(StringBeginsWith(aUri, NS_LITERAL_STRING("urn:moz-tts:osx:")),
"OSXSpeechSynthesizerService doesn't allow this voice URI");
NSSpeechSynthesizer* synth = [[NSSpeechSynthesizer alloc] init];
// strlen("urn:moz-tts:osx:") == 16
NSString* identifier = nsCocoaUtils::ToNSString(Substring(aUri, 16));
[synth setVoice:identifier];
// default rate is 180-220
[synth setObject:[NSNumber numberWithInt:aRate * 200]
forProperty:NSSpeechRateProperty error:nil];
// volume allows 0.0-1.0
[synth setObject:[NSNumber numberWithFloat:aVolume]
forProperty:NSSpeechVolumeProperty error:nil];
// Use default pitch value to calculate this
NSNumber* defaultPitch =
[synth objectForProperty:NSSpeechPitchBaseProperty error:nil];
if (defaultPitch) {
int newPitch = [defaultPitch intValue] * (aPitch / 2 + 0.5);
[synth setObject:[NSNumber numberWithInt:newPitch]
forProperty:NSSpeechPitchBaseProperty error:nil];
}
nsAutoString escapedText;
// We need to map the the offsets from the given text to the escaped text.
// The index of the offsets array is the position in the escaped text,
// the element value is the position in the user-supplied text.
nsTArray<size_t> offsets;
offsets.SetCapacity(aText.Length());
// This loop looks for occurances of "[[" or "]]", escapes them, and
// populates the offsets array to supply a map to the original offsets.
for (size_t i = 0; i < aText.Length(); i++) {
if (aText.Length() > i + 1 &&
((aText[i] == ']' && aText[i+1] == ']') ||
(aText[i] == '[' && aText[i+1] == '['))) {
escapedText.AppendLiteral(DLIM_ESCAPE_START);
offsets.AppendElements(strlen(DLIM_ESCAPE_START));
escapedText.Append(aText[i]);
offsets.AppendElement(i);
escapedText.Append(aText[++i]);
offsets.AppendElement(i);
escapedText.AppendLiteral(DLIM_ESCAPE_END);
offsets.AppendElements(strlen(DLIM_ESCAPE_END));
} else {
escapedText.Append(aText[i]);
offsets.AppendElement(i);
}
}
RefPtr<SpeechTaskCallback> callback = new SpeechTaskCallback(aTask, synth, offsets);
nsresult rv = aTask->Setup(callback, 0, 0, 0);
NS_ENSURE_SUCCESS(rv, rv);
SpeechDelegate* delegate = [[SpeechDelegate alloc] initWithCallback:callback];
[synth setDelegate:delegate];
[delegate release ];
NSString* text = nsCocoaUtils::ToNSString(escapedText);
BOOL success = [synth startSpeakingString:text];
NS_ENSURE_TRUE(success, NS_ERROR_FAILURE);
aTask->DispatchStart();
return NS_OK;
NS_OBJC_END_TRY_ABORT_BLOCK_NSRESULT;
}
NS_IMETHODIMP
OSXSpeechSynthesizerService::GetServiceType(SpeechServiceType* aServiceType)
{
*aServiceType = nsISpeechService::SERVICETYPE_INDIRECT_AUDIO;
return NS_OK;
}
NS_IMETHODIMP
OSXSpeechSynthesizerService::Observe(nsISupports* aSubject, const char* aTopic,
const char16_t* aData)
{
return NS_OK;
}
OSXSpeechSynthesizerService*
OSXSpeechSynthesizerService::GetInstance()
{
MOZ_ASSERT(NS_IsMainThread());
if (XRE_GetProcessType() != GeckoProcessType_Default) {
return nullptr;
}
if (!sSingleton) {
RefPtr<OSXSpeechSynthesizerService> speechService =
new OSXSpeechSynthesizerService();
if (speechService->Init()) {
sSingleton = speechService;
}
}
return sSingleton;
}
already_AddRefed<OSXSpeechSynthesizerService>
OSXSpeechSynthesizerService::GetInstanceForService()
{
RefPtr<OSXSpeechSynthesizerService> speechService = GetInstance();
return speechService.forget();
}
void
OSXSpeechSynthesizerService::Shutdown()
{
if (!sSingleton) {
return;
}
sSingleton = nullptr;
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,12 @@
# -*- Mode: python; indent-tabs-mode: nil; tab-width: 40 -*-
# vim: set filetype=python:
# This Source Code Form is subject to the terms of the Mozilla Public
# License, v. 2.0. If a copy of the MPL was not distributed with this
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
SOURCES += [
'OSXSpeechSynthesizerModule.cpp',
'OSXSpeechSynthesizerService.mm'
]
FINAL_LIBRARY = 'xul'

View file

@ -0,0 +1,32 @@
<!DOCTYPE html>
<html class="reftest-wait">
<head>
<meta charset="utf-8">
<script type="application/javascript">
function f()
{
if (speechSynthesis.getVoices().length == 0) {
// No synthesis backend to test this
document.documentElement.removeAttribute('class');
return;
}
var s = new SpeechSynthesisUtterance("hello world");
s.onerror = () => {
// No synthesis backend to test this
document.documentElement.removeAttribute('class');
return;
}
s.onend = () => {
document.documentElement.removeAttribute('class');
};
speechSynthesis.speak(s);
speechSynthesis.cancel();
speechSynthesis.pause();
speechSynthesis.resume();
}
</script>
</head>
<body onload="f();">
</body>
</html>

View file

@ -0,0 +1 @@
skip-if(!cocoaWidget) pref(media.webspeech.synth.enabled,true) load 1230428.html # bug 1230428

View file

@ -0,0 +1,49 @@
/* -*- Mode: c++; c-basic-offset: 2; indent-tabs-mode: nil; tab-width: 40 -*- */
/* vim: set ts=2 et sw=2 tw=80: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
* You can obtain one at http://mozilla.org/MPL/2.0/. */
include protocol PContent;
include protocol PSpeechSynthesisRequest;
namespace mozilla {
namespace dom {
struct RemoteVoice {
nsString voiceURI;
nsString name;
nsString lang;
bool localService;
bool queued;
};
sync protocol PSpeechSynthesis
{
manager PContent;
manages PSpeechSynthesisRequest;
child:
async VoiceAdded(RemoteVoice aVoice);
async VoiceRemoved(nsString aUri);
async SetDefaultVoice(nsString aUri, bool aIsDefault);
async IsSpeakingChanged(bool aIsSpeaking);
async NotifyVoicesChanged();
parent:
async __delete__();
async PSpeechSynthesisRequest(nsString aText, nsString aUri, nsString aLang,
float aVolume, float aRate, float aPitch);
sync ReadVoicesAndState() returns (RemoteVoice[] aVoices,
nsString[] aDefaults, bool aIsSpeaking);
};
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,46 @@
/* -*- Mode: c++; c-basic-offset: 2; indent-tabs-mode: nil; tab-width: 40 -*- */
/* vim: set ts=2 et sw=2 tw=80: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
* You can obtain one at http://mozilla.org/MPL/2.0/. */
include protocol PSpeechSynthesis;
namespace mozilla {
namespace dom {
async protocol PSpeechSynthesisRequest
{
manager PSpeechSynthesis;
parent:
async __delete__();
async Pause();
async Resume();
async Cancel();
async ForceEnd();
async SetAudioOutputVolume(float aVolume);
child:
async OnEnd(bool aIsError, float aElapsedTime, uint32_t aCharIndex);
async OnStart(nsString aUri);
async OnPause(float aElapsedTime, uint32_t aCharIndex);
async OnResume(float aElapsedTime, uint32_t aCharIndex);
async OnBoundary(nsString aName, float aElapsedTime, uint32_t aCharIndex);
async OnMark(nsString aName, float aElapsedTime, uint32_t aCharIndex);
};
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,213 @@
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
* You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "SpeechSynthesisChild.h"
#include "nsSynthVoiceRegistry.h"
namespace mozilla {
namespace dom {
SpeechSynthesisChild::SpeechSynthesisChild()
{
MOZ_COUNT_CTOR(SpeechSynthesisChild);
}
SpeechSynthesisChild::~SpeechSynthesisChild()
{
MOZ_COUNT_DTOR(SpeechSynthesisChild);
}
bool
SpeechSynthesisChild::RecvVoiceAdded(const RemoteVoice& aVoice)
{
nsSynthVoiceRegistry::RecvAddVoice(aVoice);
return true;
}
bool
SpeechSynthesisChild::RecvVoiceRemoved(const nsString& aUri)
{
nsSynthVoiceRegistry::RecvRemoveVoice(aUri);
return true;
}
bool
SpeechSynthesisChild::RecvSetDefaultVoice(const nsString& aUri,
const bool& aIsDefault)
{
nsSynthVoiceRegistry::RecvSetDefaultVoice(aUri, aIsDefault);
return true;
}
bool
SpeechSynthesisChild::RecvIsSpeakingChanged(const bool& aIsSpeaking)
{
nsSynthVoiceRegistry::RecvIsSpeakingChanged(aIsSpeaking);
return true;
}
bool
SpeechSynthesisChild::RecvNotifyVoicesChanged()
{
nsSynthVoiceRegistry::RecvNotifyVoicesChanged();
return true;
}
PSpeechSynthesisRequestChild*
SpeechSynthesisChild::AllocPSpeechSynthesisRequestChild(const nsString& aText,
const nsString& aLang,
const nsString& aUri,
const float& aVolume,
const float& aRate,
const float& aPitch)
{
MOZ_CRASH("Caller is supposed to manually construct a request!");
}
bool
SpeechSynthesisChild::DeallocPSpeechSynthesisRequestChild(PSpeechSynthesisRequestChild* aActor)
{
delete aActor;
return true;
}
// SpeechSynthesisRequestChild
SpeechSynthesisRequestChild::SpeechSynthesisRequestChild(SpeechTaskChild* aTask)
: mTask(aTask)
{
mTask->mActor = this;
MOZ_COUNT_CTOR(SpeechSynthesisRequestChild);
}
SpeechSynthesisRequestChild::~SpeechSynthesisRequestChild()
{
MOZ_COUNT_DTOR(SpeechSynthesisRequestChild);
}
bool
SpeechSynthesisRequestChild::RecvOnStart(const nsString& aUri)
{
mTask->DispatchStartImpl(aUri);
return true;
}
bool
SpeechSynthesisRequestChild::RecvOnEnd(const bool& aIsError,
const float& aElapsedTime,
const uint32_t& aCharIndex)
{
SpeechSynthesisRequestChild* actor = mTask->mActor;
mTask->mActor = nullptr;
if (aIsError) {
mTask->DispatchErrorImpl(aElapsedTime, aCharIndex);
} else {
mTask->DispatchEndImpl(aElapsedTime, aCharIndex);
}
actor->Send__delete__(actor);
return true;
}
bool
SpeechSynthesisRequestChild::RecvOnPause(const float& aElapsedTime,
const uint32_t& aCharIndex)
{
mTask->DispatchPauseImpl(aElapsedTime, aCharIndex);
return true;
}
bool
SpeechSynthesisRequestChild::RecvOnResume(const float& aElapsedTime,
const uint32_t& aCharIndex)
{
mTask->DispatchResumeImpl(aElapsedTime, aCharIndex);
return true;
}
bool
SpeechSynthesisRequestChild::RecvOnBoundary(const nsString& aName,
const float& aElapsedTime,
const uint32_t& aCharIndex)
{
mTask->DispatchBoundaryImpl(aName, aElapsedTime, aCharIndex);
return true;
}
bool
SpeechSynthesisRequestChild::RecvOnMark(const nsString& aName,
const float& aElapsedTime,
const uint32_t& aCharIndex)
{
mTask->DispatchMarkImpl(aName, aElapsedTime, aCharIndex);
return true;
}
// SpeechTaskChild
SpeechTaskChild::SpeechTaskChild(SpeechSynthesisUtterance* aUtterance)
: nsSpeechTask(aUtterance)
{
}
NS_IMETHODIMP
SpeechTaskChild::Setup(nsISpeechTaskCallback* aCallback,
uint32_t aChannels, uint32_t aRate, uint8_t argc)
{
MOZ_CRASH("Should never be called from child");
}
NS_IMETHODIMP
SpeechTaskChild::SendAudio(JS::Handle<JS::Value> aData, JS::Handle<JS::Value> aLandmarks,
JSContext* aCx)
{
MOZ_CRASH("Should never be called from child");
}
NS_IMETHODIMP
SpeechTaskChild::SendAudioNative(int16_t* aData, uint32_t aDataLen)
{
MOZ_CRASH("Should never be called from child");
}
void
SpeechTaskChild::Pause()
{
MOZ_ASSERT(mActor);
mActor->SendPause();
}
void
SpeechTaskChild::Resume()
{
MOZ_ASSERT(mActor);
mActor->SendResume();
}
void
SpeechTaskChild::Cancel()
{
MOZ_ASSERT(mActor);
mActor->SendCancel();
}
void
SpeechTaskChild::ForceEnd()
{
MOZ_ASSERT(mActor);
mActor->SendForceEnd();
}
void
SpeechTaskChild::SetAudioOutputVolume(float aVolume)
{
if (mActor) {
mActor->SendSetAudioOutputVolume(aVolume);
}
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,106 @@
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
* You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechSynthesisChild_h
#define mozilla_dom_SpeechSynthesisChild_h
#include "mozilla/Attributes.h"
#include "mozilla/dom/PSpeechSynthesisChild.h"
#include "mozilla/dom/PSpeechSynthesisRequestChild.h"
#include "nsSpeechTask.h"
namespace mozilla {
namespace dom {
class nsSynthVoiceRegistry;
class SpeechSynthesisRequestChild;
class SpeechTaskChild;
class SpeechSynthesisChild : public PSpeechSynthesisChild
{
friend class nsSynthVoiceRegistry;
public:
bool RecvVoiceAdded(const RemoteVoice& aVoice) override;
bool RecvVoiceRemoved(const nsString& aUri) override;
bool RecvSetDefaultVoice(const nsString& aUri, const bool& aIsDefault) override;
bool RecvIsSpeakingChanged(const bool& aIsSpeaking) override;
bool RecvNotifyVoicesChanged() override;
protected:
SpeechSynthesisChild();
virtual ~SpeechSynthesisChild();
PSpeechSynthesisRequestChild* AllocPSpeechSynthesisRequestChild(const nsString& aLang,
const nsString& aUri,
const nsString& aText,
const float& aVolume,
const float& aPitch,
const float& aRate) override;
bool DeallocPSpeechSynthesisRequestChild(PSpeechSynthesisRequestChild* aActor) override;
};
class SpeechSynthesisRequestChild : public PSpeechSynthesisRequestChild
{
public:
explicit SpeechSynthesisRequestChild(SpeechTaskChild* aTask);
virtual ~SpeechSynthesisRequestChild();
protected:
bool RecvOnStart(const nsString& aUri) override;
bool RecvOnEnd(const bool& aIsError,
const float& aElapsedTime,
const uint32_t& aCharIndex) override;
bool RecvOnPause(const float& aElapsedTime, const uint32_t& aCharIndex) override;
bool RecvOnResume(const float& aElapsedTime, const uint32_t& aCharIndex) override;
bool RecvOnBoundary(const nsString& aName, const float& aElapsedTime,
const uint32_t& aCharIndex) override;
bool RecvOnMark(const nsString& aName, const float& aElapsedTime,
const uint32_t& aCharIndex) override;
RefPtr<SpeechTaskChild> mTask;
};
class SpeechTaskChild : public nsSpeechTask
{
friend class SpeechSynthesisRequestChild;
public:
explicit SpeechTaskChild(SpeechSynthesisUtterance* aUtterance);
NS_IMETHOD Setup(nsISpeechTaskCallback* aCallback,
uint32_t aChannels, uint32_t aRate, uint8_t argc) override;
NS_IMETHOD SendAudio(JS::Handle<JS::Value> aData, JS::Handle<JS::Value> aLandmarks,
JSContext* aCx) override;
NS_IMETHOD SendAudioNative(int16_t* aData, uint32_t aDataLen) override;
void Pause() override;
void Resume() override;
void Cancel() override;
void ForceEnd() override;
void SetAudioOutputVolume(float aVolume) override;
private:
SpeechSynthesisRequestChild* mActor;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,234 @@
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
* You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "SpeechSynthesisParent.h"
#include "nsSynthVoiceRegistry.h"
namespace mozilla {
namespace dom {
SpeechSynthesisParent::SpeechSynthesisParent()
{
MOZ_COUNT_CTOR(SpeechSynthesisParent);
}
SpeechSynthesisParent::~SpeechSynthesisParent()
{
MOZ_COUNT_DTOR(SpeechSynthesisParent);
}
void
SpeechSynthesisParent::ActorDestroy(ActorDestroyReason aWhy)
{
// Implement me! Bug 1005141
}
bool
SpeechSynthesisParent::RecvReadVoicesAndState(InfallibleTArray<RemoteVoice>* aVoices,
InfallibleTArray<nsString>* aDefaults,
bool* aIsSpeaking)
{
nsSynthVoiceRegistry::GetInstance()->SendVoicesAndState(aVoices, aDefaults,
aIsSpeaking);
return true;
}
PSpeechSynthesisRequestParent*
SpeechSynthesisParent::AllocPSpeechSynthesisRequestParent(const nsString& aText,
const nsString& aLang,
const nsString& aUri,
const float& aVolume,
const float& aRate,
const float& aPitch)
{
RefPtr<SpeechTaskParent> task = new SpeechTaskParent(aVolume, aText);
SpeechSynthesisRequestParent* actor = new SpeechSynthesisRequestParent(task);
return actor;
}
bool
SpeechSynthesisParent::DeallocPSpeechSynthesisRequestParent(PSpeechSynthesisRequestParent* aActor)
{
delete aActor;
return true;
}
bool
SpeechSynthesisParent::RecvPSpeechSynthesisRequestConstructor(PSpeechSynthesisRequestParent* aActor,
const nsString& aText,
const nsString& aLang,
const nsString& aUri,
const float& aVolume,
const float& aRate,
const float& aPitch)
{
MOZ_ASSERT(aActor);
SpeechSynthesisRequestParent* actor =
static_cast<SpeechSynthesisRequestParent*>(aActor);
nsSynthVoiceRegistry::GetInstance()->Speak(aText, aLang, aUri, aVolume, aRate,
aPitch, actor->mTask);
return true;
}
// SpeechSynthesisRequestParent
SpeechSynthesisRequestParent::SpeechSynthesisRequestParent(SpeechTaskParent* aTask)
: mTask(aTask)
{
mTask->mActor = this;
MOZ_COUNT_CTOR(SpeechSynthesisRequestParent);
}
SpeechSynthesisRequestParent::~SpeechSynthesisRequestParent()
{
if (mTask) {
mTask->mActor = nullptr;
// If we still have a task, cancel it.
mTask->Cancel();
}
MOZ_COUNT_DTOR(SpeechSynthesisRequestParent);
}
void
SpeechSynthesisRequestParent::ActorDestroy(ActorDestroyReason aWhy)
{
// Implement me! Bug 1005141
}
bool
SpeechSynthesisRequestParent::RecvPause()
{
MOZ_ASSERT(mTask);
mTask->Pause();
return true;
}
bool
SpeechSynthesisRequestParent::Recv__delete__()
{
MOZ_ASSERT(mTask);
mTask->mActor = nullptr;
mTask = nullptr;
return true;
}
bool
SpeechSynthesisRequestParent::RecvResume()
{
MOZ_ASSERT(mTask);
mTask->Resume();
return true;
}
bool
SpeechSynthesisRequestParent::RecvCancel()
{
MOZ_ASSERT(mTask);
mTask->Cancel();
return true;
}
bool
SpeechSynthesisRequestParent::RecvForceEnd()
{
MOZ_ASSERT(mTask);
mTask->ForceEnd();
return true;
}
bool
SpeechSynthesisRequestParent::RecvSetAudioOutputVolume(const float& aVolume)
{
MOZ_ASSERT(mTask);
mTask->SetAudioOutputVolume(aVolume);
return true;
}
// SpeechTaskParent
nsresult
SpeechTaskParent::DispatchStartImpl(const nsAString& aUri)
{
MOZ_ASSERT(mActor);
if(NS_WARN_IF(!(mActor->SendOnStart(nsString(aUri))))) {
return NS_ERROR_FAILURE;
}
return NS_OK;
}
nsresult
SpeechTaskParent::DispatchEndImpl(float aElapsedTime, uint32_t aCharIndex)
{
if (!mActor) {
// Child is already gone.
return NS_OK;
}
if(NS_WARN_IF(!(mActor->SendOnEnd(false, aElapsedTime, aCharIndex)))) {
return NS_ERROR_FAILURE;
}
return NS_OK;
}
nsresult
SpeechTaskParent::DispatchPauseImpl(float aElapsedTime, uint32_t aCharIndex)
{
MOZ_ASSERT(mActor);
if(NS_WARN_IF(!(mActor->SendOnPause(aElapsedTime, aCharIndex)))) {
return NS_ERROR_FAILURE;
}
return NS_OK;
}
nsresult
SpeechTaskParent::DispatchResumeImpl(float aElapsedTime, uint32_t aCharIndex)
{
MOZ_ASSERT(mActor);
if(NS_WARN_IF(!(mActor->SendOnResume(aElapsedTime, aCharIndex)))) {
return NS_ERROR_FAILURE;
}
return NS_OK;
}
nsresult
SpeechTaskParent::DispatchErrorImpl(float aElapsedTime, uint32_t aCharIndex)
{
MOZ_ASSERT(mActor);
if(NS_WARN_IF(!(mActor->SendOnEnd(true, aElapsedTime, aCharIndex)))) {
return NS_ERROR_FAILURE;
}
return NS_OK;
}
nsresult
SpeechTaskParent::DispatchBoundaryImpl(const nsAString& aName,
float aElapsedTime, uint32_t aCharIndex)
{
MOZ_ASSERT(mActor);
if(NS_WARN_IF(!(mActor->SendOnBoundary(nsString(aName), aElapsedTime, aCharIndex)))) {
return NS_ERROR_FAILURE;
}
return NS_OK;
}
nsresult
SpeechTaskParent::DispatchMarkImpl(const nsAString& aName,
float aElapsedTime, uint32_t aCharIndex)
{
MOZ_ASSERT(mActor);
if(NS_WARN_IF(!(mActor->SendOnMark(nsString(aName), aElapsedTime, aCharIndex)))) {
return NS_ERROR_FAILURE;
}
return NS_OK;
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,108 @@
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
* You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechSynthesisParent_h
#define mozilla_dom_SpeechSynthesisParent_h
#include "mozilla/dom/PSpeechSynthesisParent.h"
#include "mozilla/dom/PSpeechSynthesisRequestParent.h"
#include "nsSpeechTask.h"
namespace mozilla {
namespace dom {
class ContentParent;
class SpeechTaskParent;
class SpeechSynthesisRequestParent;
class SpeechSynthesisParent : public PSpeechSynthesisParent
{
friend class ContentParent;
friend class SpeechSynthesisRequestParent;
public:
void ActorDestroy(ActorDestroyReason aWhy) override;
bool RecvReadVoicesAndState(InfallibleTArray<RemoteVoice>* aVoices,
InfallibleTArray<nsString>* aDefaults,
bool* aIsSpeaking) override;
protected:
SpeechSynthesisParent();
virtual ~SpeechSynthesisParent();
PSpeechSynthesisRequestParent* AllocPSpeechSynthesisRequestParent(const nsString& aText,
const nsString& aLang,
const nsString& aUri,
const float& aVolume,
const float& aRate,
const float& aPitch)
override;
bool DeallocPSpeechSynthesisRequestParent(PSpeechSynthesisRequestParent* aActor) override;
bool RecvPSpeechSynthesisRequestConstructor(PSpeechSynthesisRequestParent* aActor,
const nsString& aText,
const nsString& aLang,
const nsString& aUri,
const float& aVolume,
const float& aRate,
const float& aPitch) override;
};
class SpeechSynthesisRequestParent : public PSpeechSynthesisRequestParent
{
public:
explicit SpeechSynthesisRequestParent(SpeechTaskParent* aTask);
virtual ~SpeechSynthesisRequestParent();
RefPtr<SpeechTaskParent> mTask;
protected:
void ActorDestroy(ActorDestroyReason aWhy) override;
bool RecvPause() override;
bool RecvResume() override;
bool RecvCancel() override;
bool RecvForceEnd() override;
bool RecvSetAudioOutputVolume(const float& aVolume) override;
bool Recv__delete__() override;
};
class SpeechTaskParent : public nsSpeechTask
{
friend class SpeechSynthesisRequestParent;
public:
SpeechTaskParent(float aVolume, const nsAString& aUtterance)
: nsSpeechTask(aVolume, aUtterance) {}
nsresult DispatchStartImpl(const nsAString& aUri);
nsresult DispatchEndImpl(float aElapsedTime, uint32_t aCharIndex);
nsresult DispatchPauseImpl(float aElapsedTime, uint32_t aCharIndex);
nsresult DispatchResumeImpl(float aElapsedTime, uint32_t aCharIndex);
nsresult DispatchErrorImpl(float aElapsedTime, uint32_t aCharIndex);
nsresult DispatchBoundaryImpl(const nsAString& aName,
float aElapsedTime, uint32_t aCharIndex);
nsresult DispatchMarkImpl(const nsAString& aName,
float aElapsedTime, uint32_t aCharIndex);
private:
SpeechSynthesisRequestParent* mActor;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,70 @@
# vim: set filetype=python:
# This Source Code Form is subject to the terms of the Mozilla Public
# License, v. 2.0. If a copy of the MPL was not distributed with this
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
if CONFIG['MOZ_WEBSPEECH']:
MOCHITEST_MANIFESTS += [
'test/mochitest.ini',
'test/startup/mochitest.ini',
]
XPIDL_MODULE = 'dom_webspeechsynth'
XPIDL_SOURCES += [
'nsISpeechService.idl',
'nsISynthVoiceRegistry.idl'
]
EXPORTS.mozilla.dom += [
'ipc/SpeechSynthesisChild.h',
'ipc/SpeechSynthesisParent.h',
'nsSpeechTask.h',
'nsSynthVoiceRegistry.h',
'SpeechSynthesis.h',
'SpeechSynthesisUtterance.h',
'SpeechSynthesisVoice.h',
]
UNIFIED_SOURCES += [
'ipc/SpeechSynthesisChild.cpp',
'ipc/SpeechSynthesisParent.cpp',
'nsSpeechTask.cpp',
'nsSynthVoiceRegistry.cpp',
'SpeechSynthesis.cpp',
'SpeechSynthesisUtterance.cpp',
'SpeechSynthesisVoice.cpp',
]
if CONFIG['MOZ_WEBSPEECH_TEST_BACKEND']:
UNIFIED_SOURCES += [
'test/FakeSynthModule.cpp',
'test/nsFakeSynthServices.cpp'
]
if CONFIG['MOZ_WIDGET_TOOLKIT'] == 'windows':
DIRS += ['windows']
if CONFIG['MOZ_WIDGET_TOOLKIT'] == 'cocoa':
DIRS += ['cocoa']
if CONFIG['MOZ_SYNTH_SPEECHD']:
DIRS += ['speechd']
if CONFIG['MOZ_SYNTH_PICO']:
DIRS += ['pico']
IPDL_SOURCES += [
'ipc/PSpeechSynthesis.ipdl',
'ipc/PSpeechSynthesisRequest.ipdl',
]
include('/ipc/chromium/chromium-config.mozbuild')
FINAL_LIBRARY = 'xul'
LOCAL_INCLUDES += [
'ipc',
]
if CONFIG['GNU_CXX']:
CXXFLAGS += ['-Wno-error=shadow']

View file

@ -0,0 +1,173 @@
/* -*- Mode: IDL; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
* You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsISupports.idl"
typedef unsigned short SpeechServiceType;
/**
* A callback is implemented by the service. For direct audio services, it is
* required to implement these, although it could be helpful to use the
* cancel method for shutting down the speech resources.
*/
[scriptable, uuid(c576de0c-8a3d-4570-be7e-9876d3e5bed2)]
interface nsISpeechTaskCallback : nsISupports
{
/**
* The user or application has paused the speech.
*/
void onPause();
/**
* The user or application has resumed the speech.
*/
void onResume();
/**
* The user or application has canceled the speech.
*/
void onCancel();
/**
* The user or application has changed the volume of this speech.
* This is only used on indirect audio service type.
*/
void onVolumeChanged(in float aVolume);
};
/**
* A task is associated with a single utterance. It is provided by the browser
* to the service in the speak() method.
*/
[scriptable, builtinclass, uuid(ad59949c-2437-4b35-8eeb-d760caab75c5)]
interface nsISpeechTask : nsISupports
{
/**
* Prepare browser for speech.
*
* @param aCallback callback object for mid-speech operations.
* @param aChannels number of audio channels. Only required
* in direct audio services
* @param aRate audio rate. Only required in direct audio services
*/
[optional_argc] void setup(in nsISpeechTaskCallback aCallback,
[optional] in uint32_t aChannels,
[optional] in uint32_t aRate);
/**
* Send audio data to browser.
*
* @param aData an Int16Array with PCM-16 audio data.
* @param aLandmarks an array of sample offset and landmark pairs.
* Used for emiting boundary and mark events.
*/
[implicit_jscontext]
void sendAudio(in jsval aData, in jsval aLandmarks);
[noscript]
void sendAudioNative([array, size_is(aDataLen)] in short aData, in unsigned long aDataLen);
/**
* Dispatch start event.
*/
void dispatchStart();
/**
* Dispatch end event.
*
* @param aElapsedTime time in seconds since speech has started.
* @param aCharIndex offset of spoken characters.
*/
void dispatchEnd(in float aElapsedTime, in unsigned long aCharIndex);
/**
* Dispatch pause event.
*
* @param aElapsedTime time in seconds since speech has started.
* @param aCharIndex offset of spoken characters.
*/
void dispatchPause(in float aElapsedTime, in unsigned long aCharIndex);
/**
* Dispatch resume event.
*
* @param aElapsedTime time in seconds since speech has started.
* @param aCharIndex offset of spoken characters.
*/
void dispatchResume(in float aElapsedTime, in unsigned long aCharIndex);
/**
* Dispatch error event.
*
* @param aElapsedTime time in seconds since speech has started.
* @param aCharIndex offset of spoken characters.
*/
void dispatchError(in float aElapsedTime, in unsigned long aCharIndex);
/**
* Dispatch boundary event.
*
* @param aName name of boundary, 'word' or 'sentence'
* @param aElapsedTime time in seconds since speech has started.
* @param aCharIndex offset of spoken characters.
*/
void dispatchBoundary(in DOMString aName, in float aElapsedTime,
in unsigned long aCharIndex);
/**
* Dispatch mark event.
*
* @param aName mark identifier.
* @param aElapsedTime time in seconds since speech has started.
* @param aCharIndex offset of spoken characters.
*/
void dispatchMark(in DOMString aName, in float aElapsedTime, in unsigned long aCharIndex);
};
/**
* The main interface of a speech synthesis service.
*
* A service's speak method could be implemented in two ways:
* 1. Indirect audio - the service is responsible for outputting audio.
* The service calls the nsISpeechTask.dispatch* methods directly. Starting
* with dispatchStart() and ending with dispatchEnd or dispatchError().
*
* 2. Direct audio - the service provides us with PCM-16 data, and we output it.
* The service does not call the dispatch task methods directly. Instead,
* audio information is provided at setup(), and audio data is sent with
* sendAudio(). The utterance is terminated with an empty sendAudio().
*/
[scriptable, uuid(9b7d59db-88ff-43d0-b6ee-9f63d042d08f)]
interface nsISpeechService : nsISupports
{
/**
* Speak the given text using the voice identified byu the given uri. See
* W3C Speech API spec for information about pitch and rate.
* https://dvcs.w3.org/hg/speech-api/raw-file/tip/speechapi.html#utterance-attributes
*
* @param aText text to utter.
* @param aUri unique voice identifier.
* @param aVolume volume to speak voice in. Only relevant for indirect audio.
* @param aRate rate to speak voice in.
* @param aPitch pitch to speak voice in.
* @param aTask task instance for utterance, used for sending events or audio
* data back to browser.
*/
void speak(in DOMString aText, in DOMString aUri,
in float aVolume, in float aRate, in float aPitch,
in nsISpeechTask aTask);
const SpeechServiceType SERVICETYPE_DIRECT_AUDIO = 1;
const SpeechServiceType SERVICETYPE_INDIRECT_AUDIO = 2;
readonly attribute SpeechServiceType serviceType;
};
%{C++
// This is the service category speech services could use to start up as
// a component.
#define NS_SPEECH_SYNTH_STARTED "speech-synth-started"
%}

View file

@ -0,0 +1,77 @@
/* -*- Mode: IDL; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this file,
* You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsISupports.idl"
interface nsISpeechService;
[scriptable, builtinclass, uuid(5d7a0b38-77e5-4ee5-897c-ce5db9b85d44)]
interface nsISynthVoiceRegistry : nsISupports
{
/**
* Register a speech synthesis voice.
*
* @param aService the service that provides this voice.
* @param aUri a unique identifier for this voice.
* @param aName human-readable name for this voice.
* @param aLang a BCP 47 language tag.
* @param aLocalService true if service does not require network.
* @param aQueuesUtterances true if voice only speaks one utterance at a time
*/
void addVoice(in nsISpeechService aService, in DOMString aUri,
in DOMString aName, in DOMString aLang,
in boolean aLocalService, in boolean aQueuesUtterances);
/**
* Remove a speech synthesis voice.
*
* @param aService the service that was used to add the voice.
* @param aUri a unique identifier of an existing voice.
*/
void removeVoice(in nsISpeechService aService, in DOMString aUri);
/**
* Notify content of voice availability changes. This allows content
* to be notified of voice catalog changes in real time.
*/
void notifyVoicesChanged();
/**
* Set a voice as default.
*
* @param aUri a unique identifier of an existing voice.
* @param aIsDefault true if this voice should be toggled as default.
*/
void setDefaultVoice(in DOMString aUri, in boolean aIsDefault);
readonly attribute uint32_t voiceCount;
AString getVoice(in uint32_t aIndex);
bool isDefaultVoice(in DOMString aUri);
bool isLocalVoice(in DOMString aUri);
AString getVoiceLang(in DOMString aUri);
AString getVoiceName(in DOMString aUri);
};
%{C++
#define NS_SYNTHVOICEREGISTRY_CID \
{ /* {7090524d-5574-4492-a77f-d8d558ced59d} */ \
0x7090524d, \
0x5574, \
0x4492, \
{ 0xa7, 0x7f, 0xd8, 0xd5, 0x58, 0xce, 0xd5, 0x9d } \
}
#define NS_SYNTHVOICEREGISTRY_CONTRACTID \
"@mozilla.org/synth-voice-registry;1"
#define NS_SYNTHVOICEREGISTRY_CLASSNAME \
"Speech Synthesis Voice Registry"
%}

View file

@ -0,0 +1,783 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "AudioChannelAgent.h"
#include "AudioChannelService.h"
#include "AudioSegment.h"
#include "MediaStreamListener.h"
#include "nsSpeechTask.h"
#include "nsSynthVoiceRegistry.h"
#include "SharedBuffer.h"
#include "SpeechSynthesis.h"
// GetCurrentTime is defined in winbase.h as zero argument macro forwarding to
// GetTickCount() and conflicts with nsSpeechTask::GetCurrentTime().
#ifdef GetCurrentTime
#undef GetCurrentTime
#endif
#undef LOG
extern mozilla::LogModule* GetSpeechSynthLog();
#define LOG(type, msg) MOZ_LOG(GetSpeechSynthLog(), type, msg)
#define AUDIO_TRACK 1
namespace mozilla {
namespace dom {
class SynthStreamListener : public MediaStreamListener
{
public:
explicit SynthStreamListener(nsSpeechTask* aSpeechTask,
MediaStream* aStream) :
mSpeechTask(aSpeechTask),
mStream(aStream),
mStarted(false)
{
}
void DoNotifyStarted()
{
if (mSpeechTask) {
mSpeechTask->DispatchStartInner();
}
}
void DoNotifyFinished()
{
if (mSpeechTask) {
mSpeechTask->DispatchEndInner(mSpeechTask->GetCurrentTime(),
mSpeechTask->GetCurrentCharOffset());
}
}
void NotifyEvent(MediaStreamGraph* aGraph,
MediaStreamGraphEvent event) override
{
switch (event) {
case MediaStreamGraphEvent::EVENT_FINISHED:
{
if (!mStarted) {
mStarted = true;
nsCOMPtr<nsIRunnable> startRunnable =
NewRunnableMethod(this, &SynthStreamListener::DoNotifyStarted);
aGraph->DispatchToMainThreadAfterStreamStateUpdate(startRunnable.forget());
}
nsCOMPtr<nsIRunnable> endRunnable =
NewRunnableMethod(this, &SynthStreamListener::DoNotifyFinished);
aGraph->DispatchToMainThreadAfterStreamStateUpdate(endRunnable.forget());
}
break;
case MediaStreamGraphEvent::EVENT_REMOVED:
mSpeechTask = nullptr;
// Dereference MediaStream to destroy safety
mStream = nullptr;
break;
default:
break;
}
}
void NotifyBlockingChanged(MediaStreamGraph* aGraph, Blocking aBlocked) override
{
if (aBlocked == MediaStreamListener::UNBLOCKED && !mStarted) {
mStarted = true;
nsCOMPtr<nsIRunnable> event =
NewRunnableMethod(this, &SynthStreamListener::DoNotifyStarted);
aGraph->DispatchToMainThreadAfterStreamStateUpdate(event.forget());
}
}
private:
// Raw pointer; if we exist, the stream exists,
// and 'mSpeechTask' exclusively owns it and therefor exists as well.
nsSpeechTask* mSpeechTask;
// This is KungFuDeathGrip for MediaStream
RefPtr<MediaStream> mStream;
bool mStarted;
};
// nsSpeechTask
NS_IMPL_CYCLE_COLLECTION(nsSpeechTask, mSpeechSynthesis, mUtterance, mCallback);
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(nsSpeechTask)
NS_INTERFACE_MAP_ENTRY(nsISpeechTask)
NS_INTERFACE_MAP_ENTRY(nsIAudioChannelAgentCallback)
NS_INTERFACE_MAP_ENTRY(nsISupportsWeakReference)
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsISpeechTask)
NS_INTERFACE_MAP_END
NS_IMPL_CYCLE_COLLECTING_ADDREF(nsSpeechTask)
NS_IMPL_CYCLE_COLLECTING_RELEASE(nsSpeechTask)
nsSpeechTask::nsSpeechTask(SpeechSynthesisUtterance* aUtterance)
: mUtterance(aUtterance)
, mInited(false)
, mPrePaused(false)
, mPreCanceled(false)
, mCallback(nullptr)
, mIndirectAudio(false)
{
mText = aUtterance->mText;
mVolume = aUtterance->Volume();
}
nsSpeechTask::nsSpeechTask(float aVolume, const nsAString& aText)
: mUtterance(nullptr)
, mVolume(aVolume)
, mText(aText)
, mInited(false)
, mPrePaused(false)
, mPreCanceled(false)
, mCallback(nullptr)
, mIndirectAudio(false)
{
}
nsSpeechTask::~nsSpeechTask()
{
LOG(LogLevel::Debug, ("~nsSpeechTask"));
if (mStream) {
if (!mStream->IsDestroyed()) {
mStream->Destroy();
}
// This will finally destroyed by SynthStreamListener becasue
// MediaStream::Destroy() is async.
mStream = nullptr;
}
if (mPort) {
mPort->Destroy();
mPort = nullptr;
}
}
void
nsSpeechTask::InitDirectAudio()
{
mStream = MediaStreamGraph::GetInstance(MediaStreamGraph::AUDIO_THREAD_DRIVER,
AudioChannel::Normal)->
CreateSourceStream();
mIndirectAudio = false;
mInited = true;
}
void
nsSpeechTask::InitIndirectAudio()
{
mIndirectAudio = true;
mInited = true;
}
void
nsSpeechTask::SetChosenVoiceURI(const nsAString& aUri)
{
mChosenVoiceURI = aUri;
}
NS_IMETHODIMP
nsSpeechTask::Setup(nsISpeechTaskCallback* aCallback,
uint32_t aChannels, uint32_t aRate, uint8_t argc)
{
MOZ_ASSERT(XRE_IsParentProcess());
LOG(LogLevel::Debug, ("nsSpeechTask::Setup"));
mCallback = aCallback;
if (mIndirectAudio) {
MOZ_ASSERT(!mStream);
if (argc > 0) {
NS_WARNING("Audio info arguments in Setup() are ignored for indirect audio services.");
}
return NS_OK;
}
// mStream is set up in Init() that should be called before this.
MOZ_ASSERT(mStream);
mStream->AddListener(new SynthStreamListener(this, mStream));
// XXX: Support more than one channel
if(NS_WARN_IF(!(aChannels == 1))) {
return NS_ERROR_FAILURE;
}
mChannels = aChannels;
AudioSegment* segment = new AudioSegment();
mStream->AddAudioTrack(AUDIO_TRACK, aRate, 0, segment);
mStream->AddAudioOutput(this);
mStream->SetAudioOutputVolume(this, mVolume);
return NS_OK;
}
static RefPtr<mozilla::SharedBuffer>
makeSamples(int16_t* aData, uint32_t aDataLen)
{
RefPtr<mozilla::SharedBuffer> samples =
SharedBuffer::Create(aDataLen * sizeof(int16_t));
int16_t* frames = static_cast<int16_t*>(samples->Data());
for (uint32_t i = 0; i < aDataLen; i++) {
frames[i] = aData[i];
}
return samples;
}
NS_IMETHODIMP
nsSpeechTask::SendAudio(JS::Handle<JS::Value> aData, JS::Handle<JS::Value> aLandmarks,
JSContext* aCx)
{
MOZ_ASSERT(XRE_IsParentProcess());
if(NS_WARN_IF(!(mStream))) {
return NS_ERROR_NOT_AVAILABLE;
}
if(NS_WARN_IF(mStream->IsDestroyed())) {
return NS_ERROR_NOT_AVAILABLE;
}
if(NS_WARN_IF(!(mChannels))) {
return NS_ERROR_FAILURE;
}
if(NS_WARN_IF(!(aData.isObject()))) {
return NS_ERROR_INVALID_ARG;
}
if (mIndirectAudio) {
NS_WARNING("Can't call SendAudio from an indirect audio speech service.");
return NS_ERROR_FAILURE;
}
JS::Rooted<JSObject*> darray(aCx, &aData.toObject());
JSAutoCompartment ac(aCx, darray);
JS::Rooted<JSObject*> tsrc(aCx, nullptr);
// Allow either Int16Array or plain JS Array
if (JS_IsInt16Array(darray)) {
tsrc = darray;
} else {
bool isArray;
if (!JS_IsArrayObject(aCx, darray, &isArray)) {
return NS_ERROR_UNEXPECTED;
}
if (isArray) {
tsrc = JS_NewInt16ArrayFromArray(aCx, darray);
}
}
if (!tsrc) {
return NS_ERROR_DOM_TYPE_MISMATCH_ERR;
}
uint32_t dataLen = JS_GetTypedArrayLength(tsrc);
RefPtr<mozilla::SharedBuffer> samples;
{
JS::AutoCheckCannotGC nogc;
bool isShared;
int16_t* data = JS_GetInt16ArrayData(tsrc, &isShared, nogc);
if (isShared) {
// Must opt in to using shared data.
return NS_ERROR_DOM_TYPE_MISMATCH_ERR;
}
samples = makeSamples(data, dataLen);
}
SendAudioImpl(samples, dataLen);
return NS_OK;
}
NS_IMETHODIMP
nsSpeechTask::SendAudioNative(int16_t* aData, uint32_t aDataLen)
{
MOZ_ASSERT(XRE_IsParentProcess());
if(NS_WARN_IF(!(mStream))) {
return NS_ERROR_NOT_AVAILABLE;
}
if(NS_WARN_IF(mStream->IsDestroyed())) {
return NS_ERROR_NOT_AVAILABLE;
}
if(NS_WARN_IF(!(mChannels))) {
return NS_ERROR_FAILURE;
}
if (mIndirectAudio) {
NS_WARNING("Can't call SendAudio from an indirect audio speech service.");
return NS_ERROR_FAILURE;
}
RefPtr<mozilla::SharedBuffer> samples = makeSamples(aData, aDataLen);
SendAudioImpl(samples, aDataLen);
return NS_OK;
}
void
nsSpeechTask::SendAudioImpl(RefPtr<mozilla::SharedBuffer>& aSamples, uint32_t aDataLen)
{
if (aDataLen == 0) {
mStream->EndAllTrackAndFinish();
return;
}
AudioSegment segment;
AutoTArray<const int16_t*, 1> channelData;
channelData.AppendElement(static_cast<int16_t*>(aSamples->Data()));
segment.AppendFrames(aSamples.forget(), channelData, aDataLen,
PRINCIPAL_HANDLE_NONE);
mStream->AppendToTrack(1, &segment);
mStream->AdvanceKnownTracksTime(STREAM_TIME_MAX);
}
NS_IMETHODIMP
nsSpeechTask::DispatchStart()
{
if (!mIndirectAudio) {
NS_WARNING("Can't call DispatchStart() from a direct audio speech service");
return NS_ERROR_FAILURE;
}
return DispatchStartInner();
}
nsresult
nsSpeechTask::DispatchStartInner()
{
nsSynthVoiceRegistry::GetInstance()->SetIsSpeaking(true);
return DispatchStartImpl();
}
nsresult
nsSpeechTask::DispatchStartImpl()
{
return DispatchStartImpl(mChosenVoiceURI);
}
nsresult
nsSpeechTask::DispatchStartImpl(const nsAString& aUri)
{
LOG(LogLevel::Debug, ("nsSpeechTask::DispatchStart"));
MOZ_ASSERT(mUtterance);
if(NS_WARN_IF(!(mUtterance->mState == SpeechSynthesisUtterance::STATE_PENDING))) {
return NS_ERROR_NOT_AVAILABLE;
}
CreateAudioChannelAgent();
mUtterance->mState = SpeechSynthesisUtterance::STATE_SPEAKING;
mUtterance->mChosenVoiceURI = aUri;
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("start"), 0, 0,
EmptyString());
return NS_OK;
}
NS_IMETHODIMP
nsSpeechTask::DispatchEnd(float aElapsedTime, uint32_t aCharIndex)
{
if (!mIndirectAudio) {
NS_WARNING("Can't call DispatchEnd() from a direct audio speech service");
return NS_ERROR_FAILURE;
}
return DispatchEndInner(aElapsedTime, aCharIndex);
}
nsresult
nsSpeechTask::DispatchEndInner(float aElapsedTime, uint32_t aCharIndex)
{
if (!mPreCanceled) {
nsSynthVoiceRegistry::GetInstance()->SpeakNext();
}
return DispatchEndImpl(aElapsedTime, aCharIndex);
}
nsresult
nsSpeechTask::DispatchEndImpl(float aElapsedTime, uint32_t aCharIndex)
{
LOG(LogLevel::Debug, ("nsSpeechTask::DispatchEnd\n"));
DestroyAudioChannelAgent();
MOZ_ASSERT(mUtterance);
if(NS_WARN_IF(mUtterance->mState == SpeechSynthesisUtterance::STATE_ENDED)) {
return NS_ERROR_NOT_AVAILABLE;
}
// XXX: This should not be here, but it prevents a crash in MSG.
if (mStream) {
mStream->Destroy();
}
RefPtr<SpeechSynthesisUtterance> utterance = mUtterance;
if (mSpeechSynthesis) {
mSpeechSynthesis->OnEnd(this);
}
if (utterance->mState == SpeechSynthesisUtterance::STATE_PENDING) {
utterance->mState = SpeechSynthesisUtterance::STATE_NONE;
} else {
utterance->mState = SpeechSynthesisUtterance::STATE_ENDED;
utterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("end"),
aCharIndex, aElapsedTime,
EmptyString());
}
return NS_OK;
}
NS_IMETHODIMP
nsSpeechTask::DispatchPause(float aElapsedTime, uint32_t aCharIndex)
{
if (!mIndirectAudio) {
NS_WARNING("Can't call DispatchPause() from a direct audio speech service");
return NS_ERROR_FAILURE;
}
return DispatchPauseImpl(aElapsedTime, aCharIndex);
}
nsresult
nsSpeechTask::DispatchPauseImpl(float aElapsedTime, uint32_t aCharIndex)
{
LOG(LogLevel::Debug, ("nsSpeechTask::DispatchPause"));
MOZ_ASSERT(mUtterance);
if(NS_WARN_IF(mUtterance->mPaused)) {
return NS_ERROR_NOT_AVAILABLE;
}
if(NS_WARN_IF(mUtterance->mState == SpeechSynthesisUtterance::STATE_ENDED)) {
return NS_ERROR_NOT_AVAILABLE;
}
mUtterance->mPaused = true;
if (mUtterance->mState == SpeechSynthesisUtterance::STATE_SPEAKING) {
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("pause"),
aCharIndex, aElapsedTime,
EmptyString());
}
return NS_OK;
}
NS_IMETHODIMP
nsSpeechTask::DispatchResume(float aElapsedTime, uint32_t aCharIndex)
{
if (!mIndirectAudio) {
NS_WARNING("Can't call DispatchResume() from a direct audio speech service");
return NS_ERROR_FAILURE;
}
return DispatchResumeImpl(aElapsedTime, aCharIndex);
}
nsresult
nsSpeechTask::DispatchResumeImpl(float aElapsedTime, uint32_t aCharIndex)
{
LOG(LogLevel::Debug, ("nsSpeechTask::DispatchResume"));
MOZ_ASSERT(mUtterance);
if(NS_WARN_IF(!(mUtterance->mPaused))) {
return NS_ERROR_NOT_AVAILABLE;
}
if(NS_WARN_IF(mUtterance->mState == SpeechSynthesisUtterance::STATE_ENDED)) {
return NS_ERROR_NOT_AVAILABLE;
}
mUtterance->mPaused = false;
if (mUtterance->mState == SpeechSynthesisUtterance::STATE_SPEAKING) {
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("resume"),
aCharIndex, aElapsedTime,
EmptyString());
}
return NS_OK;
}
NS_IMETHODIMP
nsSpeechTask::DispatchError(float aElapsedTime, uint32_t aCharIndex)
{
LOG(LogLevel::Debug, ("nsSpeechTask::DispatchError"));
if (!mIndirectAudio) {
NS_WARNING("Can't call DispatchError() from a direct audio speech service");
return NS_ERROR_FAILURE;
}
if (!mPreCanceled) {
nsSynthVoiceRegistry::GetInstance()->SpeakNext();
}
return DispatchErrorImpl(aElapsedTime, aCharIndex);
}
nsresult
nsSpeechTask::DispatchErrorImpl(float aElapsedTime, uint32_t aCharIndex)
{
MOZ_ASSERT(mUtterance);
if(NS_WARN_IF(mUtterance->mState == SpeechSynthesisUtterance::STATE_ENDED)) {
return NS_ERROR_NOT_AVAILABLE;
}
if (mSpeechSynthesis) {
mSpeechSynthesis->OnEnd(this);
}
mUtterance->mState = SpeechSynthesisUtterance::STATE_ENDED;
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("error"),
aCharIndex, aElapsedTime,
EmptyString());
return NS_OK;
}
NS_IMETHODIMP
nsSpeechTask::DispatchBoundary(const nsAString& aName,
float aElapsedTime, uint32_t aCharIndex)
{
if (!mIndirectAudio) {
NS_WARNING("Can't call DispatchBoundary() from a direct audio speech service");
return NS_ERROR_FAILURE;
}
return DispatchBoundaryImpl(aName, aElapsedTime, aCharIndex);
}
nsresult
nsSpeechTask::DispatchBoundaryImpl(const nsAString& aName,
float aElapsedTime, uint32_t aCharIndex)
{
MOZ_ASSERT(mUtterance);
if(NS_WARN_IF(!(mUtterance->mState == SpeechSynthesisUtterance::STATE_SPEAKING))) {
return NS_ERROR_NOT_AVAILABLE;
}
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("boundary"),
aCharIndex, aElapsedTime,
aName);
return NS_OK;
}
NS_IMETHODIMP
nsSpeechTask::DispatchMark(const nsAString& aName,
float aElapsedTime, uint32_t aCharIndex)
{
if (!mIndirectAudio) {
NS_WARNING("Can't call DispatchMark() from a direct audio speech service");
return NS_ERROR_FAILURE;
}
return DispatchMarkImpl(aName, aElapsedTime, aCharIndex);
}
nsresult
nsSpeechTask::DispatchMarkImpl(const nsAString& aName,
float aElapsedTime, uint32_t aCharIndex)
{
MOZ_ASSERT(mUtterance);
if(NS_WARN_IF(!(mUtterance->mState == SpeechSynthesisUtterance::STATE_SPEAKING))) {
return NS_ERROR_NOT_AVAILABLE;
}
mUtterance->DispatchSpeechSynthesisEvent(NS_LITERAL_STRING("mark"),
aCharIndex, aElapsedTime,
aName);
return NS_OK;
}
void
nsSpeechTask::Pause()
{
MOZ_ASSERT(XRE_IsParentProcess());
if (mCallback) {
DebugOnly<nsresult> rv = mCallback->OnPause();
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv), "Unable to call onPause() callback");
}
if (mStream) {
mStream->Suspend();
}
if (!mInited) {
mPrePaused = true;
}
if (!mIndirectAudio) {
DispatchPauseImpl(GetCurrentTime(), GetCurrentCharOffset());
}
}
void
nsSpeechTask::Resume()
{
MOZ_ASSERT(XRE_IsParentProcess());
if (mCallback) {
DebugOnly<nsresult> rv = mCallback->OnResume();
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv),
"Unable to call onResume() callback");
}
if (mStream) {
mStream->Resume();
}
if (mPrePaused) {
mPrePaused = false;
nsSynthVoiceRegistry::GetInstance()->ResumeQueue();
}
if (!mIndirectAudio) {
DispatchResumeImpl(GetCurrentTime(), GetCurrentCharOffset());
}
}
void
nsSpeechTask::Cancel()
{
MOZ_ASSERT(XRE_IsParentProcess());
LOG(LogLevel::Debug, ("nsSpeechTask::Cancel"));
if (mCallback) {
DebugOnly<nsresult> rv = mCallback->OnCancel();
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv),
"Unable to call onCancel() callback");
}
if (mStream) {
mStream->Suspend();
}
if (!mInited) {
mPreCanceled = true;
}
if (!mIndirectAudio) {
DispatchEndInner(GetCurrentTime(), GetCurrentCharOffset());
}
}
void
nsSpeechTask::ForceEnd()
{
if (mStream) {
mStream->Suspend();
}
if (!mInited) {
mPreCanceled = true;
}
DispatchEndInner(GetCurrentTime(), GetCurrentCharOffset());
}
float
nsSpeechTask::GetCurrentTime()
{
return mStream ? (float)(mStream->GetCurrentTime() / 1000000.0) : 0;
}
uint32_t
nsSpeechTask::GetCurrentCharOffset()
{
return mStream && mStream->IsFinished() ? mText.Length() : 0;
}
void
nsSpeechTask::SetSpeechSynthesis(SpeechSynthesis* aSpeechSynthesis)
{
mSpeechSynthesis = aSpeechSynthesis;
}
void
nsSpeechTask::CreateAudioChannelAgent()
{
if (!mUtterance) {
return;
}
if (mAudioChannelAgent) {
mAudioChannelAgent->NotifyStoppedPlaying();
}
mAudioChannelAgent = new AudioChannelAgent();
mAudioChannelAgent->InitWithWeakCallback(mUtterance->GetOwner(),
static_cast<int32_t>(AudioChannelService::GetDefaultAudioChannel()),
this);
AudioPlaybackConfig config;
nsresult rv = mAudioChannelAgent->NotifyStartedPlaying(&config,
AudioChannelService::AudibleState::eAudible);
if (NS_WARN_IF(NS_FAILED(rv))) {
return;
}
WindowVolumeChanged(config.mVolume, config.mMuted);
WindowSuspendChanged(config.mSuspend);
}
void
nsSpeechTask::DestroyAudioChannelAgent()
{
if (mAudioChannelAgent) {
mAudioChannelAgent->NotifyStoppedPlaying();
mAudioChannelAgent = nullptr;
}
}
NS_IMETHODIMP
nsSpeechTask::WindowVolumeChanged(float aVolume, bool aMuted)
{
SetAudioOutputVolume(aMuted ? 0.0 : mVolume * aVolume);
return NS_OK;
}
NS_IMETHODIMP
nsSpeechTask::WindowSuspendChanged(nsSuspendedTypes aSuspend)
{
if (!mUtterance) {
return NS_OK;
}
if (aSuspend == nsISuspendedTypes::NONE_SUSPENDED &&
mUtterance->mPaused) {
Resume();
} else if (aSuspend != nsISuspendedTypes::NONE_SUSPENDED &&
!mUtterance->mPaused) {
Pause();
}
return NS_OK;
}
NS_IMETHODIMP
nsSpeechTask::WindowAudioCaptureChanged(bool aCapture)
{
// This is not supported yet.
return NS_OK;
}
void
nsSpeechTask::SetAudioOutputVolume(float aVolume)
{
if (mStream && !mStream->IsDestroyed()) {
mStream->SetAudioOutputVolume(this, aVolume);
}
if (mIndirectAudio) {
mCallback->OnVolumeChanged(aVolume);
}
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,139 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_nsSpeechTask_h
#define mozilla_dom_nsSpeechTask_h
#include "MediaStreamGraph.h"
#include "SpeechSynthesisUtterance.h"
#include "nsIAudioChannelAgent.h"
#include "nsISpeechService.h"
namespace mozilla {
class SharedBuffer;
namespace dom {
class SpeechSynthesisUtterance;
class SpeechSynthesis;
class SynthStreamListener;
class nsSpeechTask : public nsISpeechTask
, public nsIAudioChannelAgentCallback
, public nsSupportsWeakReference
{
friend class SynthStreamListener;
public:
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
NS_DECL_CYCLE_COLLECTION_CLASS_AMBIGUOUS(nsSpeechTask, nsISpeechTask)
NS_DECL_NSISPEECHTASK
NS_DECL_NSIAUDIOCHANNELAGENTCALLBACK
explicit nsSpeechTask(SpeechSynthesisUtterance* aUtterance);
nsSpeechTask(float aVolume, const nsAString& aText);
virtual void Pause();
virtual void Resume();
virtual void Cancel();
virtual void ForceEnd();
float GetCurrentTime();
uint32_t GetCurrentCharOffset();
void SetSpeechSynthesis(SpeechSynthesis* aSpeechSynthesis);
void InitDirectAudio();
void InitIndirectAudio();
void SetChosenVoiceURI(const nsAString& aUri);
virtual void SetAudioOutputVolume(float aVolume);
bool IsPreCanceled()
{
return mPreCanceled;
};
bool IsPrePaused()
{
return mPrePaused;
}
protected:
virtual ~nsSpeechTask();
nsresult DispatchStartImpl();
virtual nsresult DispatchStartImpl(const nsAString& aUri);
virtual nsresult DispatchEndImpl(float aElapsedTime, uint32_t aCharIndex);
virtual nsresult DispatchPauseImpl(float aElapsedTime, uint32_t aCharIndex);
virtual nsresult DispatchResumeImpl(float aElapsedTime, uint32_t aCharIndex);
virtual nsresult DispatchErrorImpl(float aElapsedTime, uint32_t aCharIndex);
virtual nsresult DispatchBoundaryImpl(const nsAString& aName,
float aElapsedTime,
uint32_t aCharIndex);
virtual nsresult DispatchMarkImpl(const nsAString& aName,
float aElapsedTime, uint32_t aCharIndex);
RefPtr<SpeechSynthesisUtterance> mUtterance;
float mVolume;
nsString mText;
bool mInited;
bool mPrePaused;
bool mPreCanceled;
private:
void End();
void SendAudioImpl(RefPtr<mozilla::SharedBuffer>& aSamples, uint32_t aDataLen);
nsresult DispatchStartInner();
nsresult DispatchEndInner(float aElapsedTime, uint32_t aCharIndex);
void CreateAudioChannelAgent();
void DestroyAudioChannelAgent();
RefPtr<SourceMediaStream> mStream;
RefPtr<MediaInputPort> mPort;
nsCOMPtr<nsISpeechTaskCallback> mCallback;
nsCOMPtr<nsIAudioChannelAgent> mAudioChannelAgent;
uint32_t mChannels;
RefPtr<SpeechSynthesis> mSpeechSynthesis;
bool mIndirectAudio;
nsString mChosenVoiceURI;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,835 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsILocaleService.h"
#include "nsISpeechService.h"
#include "nsServiceManagerUtils.h"
#include "nsCategoryManagerUtils.h"
#include "MediaPrefs.h"
#include "SpeechSynthesisUtterance.h"
#include "SpeechSynthesisVoice.h"
#include "nsSynthVoiceRegistry.h"
#include "nsSpeechTask.h"
#include "AudioChannelService.h"
#include "nsString.h"
#include "mozilla/StaticPtr.h"
#include "mozilla/dom/ContentChild.h"
#include "mozilla/dom/ContentParent.h"
#include "mozilla/Unused.h"
#include "SpeechSynthesisChild.h"
#include "SpeechSynthesisParent.h"
#undef LOG
extern mozilla::LogModule* GetSpeechSynthLog();
#define LOG(type, msg) MOZ_LOG(GetSpeechSynthLog(), type, msg)
namespace {
void
GetAllSpeechSynthActors(InfallibleTArray<mozilla::dom::SpeechSynthesisParent*>& aActors)
{
MOZ_ASSERT(NS_IsMainThread());
MOZ_ASSERT(aActors.IsEmpty());
AutoTArray<mozilla::dom::ContentParent*, 20> contentActors;
mozilla::dom::ContentParent::GetAll(contentActors);
for (uint32_t contentIndex = 0;
contentIndex < contentActors.Length();
++contentIndex) {
MOZ_ASSERT(contentActors[contentIndex]);
AutoTArray<mozilla::dom::PSpeechSynthesisParent*, 5> speechsynthActors;
contentActors[contentIndex]->ManagedPSpeechSynthesisParent(speechsynthActors);
for (uint32_t speechsynthIndex = 0;
speechsynthIndex < speechsynthActors.Length();
++speechsynthIndex) {
MOZ_ASSERT(speechsynthActors[speechsynthIndex]);
mozilla::dom::SpeechSynthesisParent* actor =
static_cast<mozilla::dom::SpeechSynthesisParent*>(speechsynthActors[speechsynthIndex]);
aActors.AppendElement(actor);
}
}
}
} // namespace
namespace mozilla {
namespace dom {
// VoiceData
class VoiceData final
{
private:
// Private destructor, to discourage deletion outside of Release():
~VoiceData() {}
public:
VoiceData(nsISpeechService* aService, const nsAString& aUri,
const nsAString& aName, const nsAString& aLang,
bool aIsLocal, bool aQueuesUtterances)
: mService(aService)
, mUri(aUri)
, mName(aName)
, mLang(aLang)
, mIsLocal(aIsLocal)
, mIsQueued(aQueuesUtterances) {}
NS_INLINE_DECL_REFCOUNTING(VoiceData)
nsCOMPtr<nsISpeechService> mService;
nsString mUri;
nsString mName;
nsString mLang;
bool mIsLocal;
bool mIsQueued;
};
// GlobalQueueItem
class GlobalQueueItem final
{
private:
// Private destructor, to discourage deletion outside of Release():
~GlobalQueueItem() {}
public:
GlobalQueueItem(VoiceData* aVoice, nsSpeechTask* aTask, const nsAString& aText,
const float& aVolume, const float& aRate, const float& aPitch)
: mVoice(aVoice)
, mTask(aTask)
, mText(aText)
, mVolume(aVolume)
, mRate(aRate)
, mPitch(aPitch) {}
NS_INLINE_DECL_REFCOUNTING(GlobalQueueItem)
RefPtr<VoiceData> mVoice;
RefPtr<nsSpeechTask> mTask;
nsString mText;
float mVolume;
float mRate;
float mPitch;
bool mIsLocal;
};
// nsSynthVoiceRegistry
static StaticRefPtr<nsSynthVoiceRegistry> gSynthVoiceRegistry;
NS_IMPL_ISUPPORTS(nsSynthVoiceRegistry, nsISynthVoiceRegistry)
nsSynthVoiceRegistry::nsSynthVoiceRegistry()
: mSpeechSynthChild(nullptr)
, mUseGlobalQueue(false)
, mIsSpeaking(false)
{
if (XRE_IsContentProcess()) {
mSpeechSynthChild = new SpeechSynthesisChild();
ContentChild::GetSingleton()->SendPSpeechSynthesisConstructor(mSpeechSynthChild);
InfallibleTArray<RemoteVoice> voices;
InfallibleTArray<nsString> defaults;
bool isSpeaking;
mSpeechSynthChild->SendReadVoicesAndState(&voices, &defaults, &isSpeaking);
for (uint32_t i = 0; i < voices.Length(); ++i) {
RemoteVoice voice = voices[i];
AddVoiceImpl(nullptr, voice.voiceURI(),
voice.name(), voice.lang(),
voice.localService(), voice.queued());
}
for (uint32_t i = 0; i < defaults.Length(); ++i) {
SetDefaultVoice(defaults[i], true);
}
mIsSpeaking = isSpeaking;
}
}
nsSynthVoiceRegistry::~nsSynthVoiceRegistry()
{
LOG(LogLevel::Debug, ("~nsSynthVoiceRegistry"));
// mSpeechSynthChild's lifecycle is managed by the Content protocol.
mSpeechSynthChild = nullptr;
mUriVoiceMap.Clear();
}
nsSynthVoiceRegistry*
nsSynthVoiceRegistry::GetInstance()
{
MOZ_ASSERT(NS_IsMainThread());
if (!gSynthVoiceRegistry) {
gSynthVoiceRegistry = new nsSynthVoiceRegistry();
if (XRE_IsParentProcess()) {
// Start up all speech synth services.
NS_CreateServicesFromCategory(NS_SPEECH_SYNTH_STARTED, nullptr,
NS_SPEECH_SYNTH_STARTED);
}
}
return gSynthVoiceRegistry;
}
already_AddRefed<nsSynthVoiceRegistry>
nsSynthVoiceRegistry::GetInstanceForService()
{
RefPtr<nsSynthVoiceRegistry> registry = GetInstance();
return registry.forget();
}
void
nsSynthVoiceRegistry::Shutdown()
{
LOG(LogLevel::Debug, ("[%s] nsSynthVoiceRegistry::Shutdown()",
(XRE_IsContentProcess()) ? "Content" : "Default"));
gSynthVoiceRegistry = nullptr;
}
void
nsSynthVoiceRegistry::SendVoicesAndState(InfallibleTArray<RemoteVoice>* aVoices,
InfallibleTArray<nsString>* aDefaults,
bool* aIsSpeaking)
{
for (uint32_t i=0; i < mVoices.Length(); ++i) {
RefPtr<VoiceData> voice = mVoices[i];
aVoices->AppendElement(RemoteVoice(voice->mUri, voice->mName, voice->mLang,
voice->mIsLocal, voice->mIsQueued));
}
for (uint32_t i=0; i < mDefaultVoices.Length(); ++i) {
aDefaults->AppendElement(mDefaultVoices[i]->mUri);
}
*aIsSpeaking = IsSpeaking();
}
void
nsSynthVoiceRegistry::RecvRemoveVoice(const nsAString& aUri)
{
// If we dont have a local instance of the registry yet, we will recieve current
// voices at contruction time.
if(!gSynthVoiceRegistry) {
return;
}
gSynthVoiceRegistry->RemoveVoice(nullptr, aUri);
}
void
nsSynthVoiceRegistry::RecvAddVoice(const RemoteVoice& aVoice)
{
// If we dont have a local instance of the registry yet, we will recieve current
// voices at contruction time.
if(!gSynthVoiceRegistry) {
return;
}
gSynthVoiceRegistry->AddVoiceImpl(nullptr, aVoice.voiceURI(),
aVoice.name(), aVoice.lang(),
aVoice.localService(), aVoice.queued());
}
void
nsSynthVoiceRegistry::RecvSetDefaultVoice(const nsAString& aUri, bool aIsDefault)
{
// If we dont have a local instance of the registry yet, we will recieve current
// voices at contruction time.
if(!gSynthVoiceRegistry) {
return;
}
gSynthVoiceRegistry->SetDefaultVoice(aUri, aIsDefault);
}
void
nsSynthVoiceRegistry::RecvIsSpeakingChanged(bool aIsSpeaking)
{
// If we dont have a local instance of the registry yet, we will get the
// speaking state on construction.
if(!gSynthVoiceRegistry) {
return;
}
gSynthVoiceRegistry->mIsSpeaking = aIsSpeaking;
}
void
nsSynthVoiceRegistry::RecvNotifyVoicesChanged()
{
// If we dont have a local instance of the registry yet, we don't care.
if(!gSynthVoiceRegistry) {
return;
}
gSynthVoiceRegistry->NotifyVoicesChanged();
}
NS_IMETHODIMP
nsSynthVoiceRegistry::AddVoice(nsISpeechService* aService,
const nsAString& aUri,
const nsAString& aName,
const nsAString& aLang,
bool aLocalService,
bool aQueuesUtterances)
{
LOG(LogLevel::Debug,
("nsSynthVoiceRegistry::AddVoice uri='%s' name='%s' lang='%s' local=%s queued=%s",
NS_ConvertUTF16toUTF8(aUri).get(), NS_ConvertUTF16toUTF8(aName).get(),
NS_ConvertUTF16toUTF8(aLang).get(),
aLocalService ? "true" : "false",
aQueuesUtterances ? "true" : "false"));
if(NS_WARN_IF(XRE_IsContentProcess())) {
return NS_ERROR_NOT_AVAILABLE;
}
return AddVoiceImpl(aService, aUri, aName, aLang, aLocalService, aQueuesUtterances);
}
NS_IMETHODIMP
nsSynthVoiceRegistry::RemoveVoice(nsISpeechService* aService,
const nsAString& aUri)
{
LOG(LogLevel::Debug,
("nsSynthVoiceRegistry::RemoveVoice uri='%s' (%s)",
NS_ConvertUTF16toUTF8(aUri).get(),
(XRE_IsContentProcess()) ? "child" : "parent"));
bool found = false;
VoiceData* retval = mUriVoiceMap.GetWeak(aUri, &found);
if(NS_WARN_IF(!(found))) {
return NS_ERROR_NOT_AVAILABLE;
}
if(NS_WARN_IF(!(aService == retval->mService))) {
return NS_ERROR_INVALID_ARG;
}
mVoices.RemoveElement(retval);
mDefaultVoices.RemoveElement(retval);
mUriVoiceMap.Remove(aUri);
if (retval->mIsQueued && !MediaPrefs::WebSpeechForceGlobal()) {
// Check if this is the last queued voice, and disable the global queue if
// it is.
bool queued = false;
for (uint32_t i = 0; i < mVoices.Length(); i++) {
VoiceData* voice = mVoices[i];
if (voice->mIsQueued) {
queued = true;
break;
}
}
if (!queued) {
mUseGlobalQueue = false;
}
}
nsTArray<SpeechSynthesisParent*> ssplist;
GetAllSpeechSynthActors(ssplist);
for (uint32_t i = 0; i < ssplist.Length(); ++i)
Unused << ssplist[i]->SendVoiceRemoved(nsString(aUri));
return NS_OK;
}
NS_IMETHODIMP
nsSynthVoiceRegistry::NotifyVoicesChanged()
{
if (XRE_IsParentProcess()) {
nsTArray<SpeechSynthesisParent*> ssplist;
GetAllSpeechSynthActors(ssplist);
for (uint32_t i = 0; i < ssplist.Length(); ++i)
Unused << ssplist[i]->SendNotifyVoicesChanged();
}
nsCOMPtr<nsIObserverService> obs = mozilla::services::GetObserverService();
if(NS_WARN_IF(!(obs))) {
return NS_ERROR_NOT_AVAILABLE;
}
obs->NotifyObservers(nullptr, "synth-voices-changed", nullptr);
return NS_OK;
}
NS_IMETHODIMP
nsSynthVoiceRegistry::SetDefaultVoice(const nsAString& aUri,
bool aIsDefault)
{
bool found = false;
VoiceData* retval = mUriVoiceMap.GetWeak(aUri, &found);
if(NS_WARN_IF(!(found))) {
return NS_ERROR_NOT_AVAILABLE;
}
mDefaultVoices.RemoveElement(retval);
LOG(LogLevel::Debug, ("nsSynthVoiceRegistry::SetDefaultVoice %s %s",
NS_ConvertUTF16toUTF8(aUri).get(),
aIsDefault ? "true" : "false"));
if (aIsDefault) {
mDefaultVoices.AppendElement(retval);
}
if (XRE_IsParentProcess()) {
nsTArray<SpeechSynthesisParent*> ssplist;
GetAllSpeechSynthActors(ssplist);
for (uint32_t i = 0; i < ssplist.Length(); ++i) {
Unused << ssplist[i]->SendSetDefaultVoice(nsString(aUri), aIsDefault);
}
}
return NS_OK;
}
NS_IMETHODIMP
nsSynthVoiceRegistry::GetVoiceCount(uint32_t* aRetval)
{
*aRetval = mVoices.Length();
return NS_OK;
}
NS_IMETHODIMP
nsSynthVoiceRegistry::GetVoice(uint32_t aIndex, nsAString& aRetval)
{
if(NS_WARN_IF(!(aIndex < mVoices.Length()))) {
return NS_ERROR_INVALID_ARG;
}
aRetval = mVoices[aIndex]->mUri;
return NS_OK;
}
NS_IMETHODIMP
nsSynthVoiceRegistry::IsDefaultVoice(const nsAString& aUri, bool* aRetval)
{
bool found;
VoiceData* voice = mUriVoiceMap.GetWeak(aUri, &found);
if(NS_WARN_IF(!(found))) {
return NS_ERROR_NOT_AVAILABLE;
}
for (int32_t i = mDefaultVoices.Length(); i > 0; ) {
VoiceData* defaultVoice = mDefaultVoices[--i];
if (voice->mLang.Equals(defaultVoice->mLang)) {
*aRetval = voice == defaultVoice;
return NS_OK;
}
}
*aRetval = false;
return NS_OK;
}
NS_IMETHODIMP
nsSynthVoiceRegistry::IsLocalVoice(const nsAString& aUri, bool* aRetval)
{
bool found;
VoiceData* voice = mUriVoiceMap.GetWeak(aUri, &found);
if(NS_WARN_IF(!(found))) {
return NS_ERROR_NOT_AVAILABLE;
}
*aRetval = voice->mIsLocal;
return NS_OK;
}
NS_IMETHODIMP
nsSynthVoiceRegistry::GetVoiceLang(const nsAString& aUri, nsAString& aRetval)
{
bool found;
VoiceData* voice = mUriVoiceMap.GetWeak(aUri, &found);
if(NS_WARN_IF(!(found))) {
return NS_ERROR_NOT_AVAILABLE;
}
aRetval = voice->mLang;
return NS_OK;
}
NS_IMETHODIMP
nsSynthVoiceRegistry::GetVoiceName(const nsAString& aUri, nsAString& aRetval)
{
bool found;
VoiceData* voice = mUriVoiceMap.GetWeak(aUri, &found);
if(NS_WARN_IF(!(found))) {
return NS_ERROR_NOT_AVAILABLE;
}
aRetval = voice->mName;
return NS_OK;
}
nsresult
nsSynthVoiceRegistry::AddVoiceImpl(nsISpeechService* aService,
const nsAString& aUri,
const nsAString& aName,
const nsAString& aLang,
bool aLocalService,
bool aQueuesUtterances)
{
bool found = false;
mUriVoiceMap.GetWeak(aUri, &found);
if(NS_WARN_IF(found)) {
return NS_ERROR_INVALID_ARG;
}
RefPtr<VoiceData> voice = new VoiceData(aService, aUri, aName, aLang,
aLocalService, aQueuesUtterances);
mVoices.AppendElement(voice);
mUriVoiceMap.Put(aUri, voice);
mUseGlobalQueue |= aQueuesUtterances;
nsTArray<SpeechSynthesisParent*> ssplist;
GetAllSpeechSynthActors(ssplist);
if (!ssplist.IsEmpty()) {
mozilla::dom::RemoteVoice ssvoice(nsString(aUri),
nsString(aName),
nsString(aLang),
aLocalService,
aQueuesUtterances);
for (uint32_t i = 0; i < ssplist.Length(); ++i) {
Unused << ssplist[i]->SendVoiceAdded(ssvoice);
}
}
return NS_OK;
}
bool
nsSynthVoiceRegistry::FindVoiceByLang(const nsAString& aLang,
VoiceData** aRetval)
{
nsAString::const_iterator dashPos, start, end;
aLang.BeginReading(start);
aLang.EndReading(end);
while (true) {
nsAutoString langPrefix(Substring(start, end));
for (int32_t i = mDefaultVoices.Length(); i > 0; ) {
VoiceData* voice = mDefaultVoices[--i];
if (StringBeginsWith(voice->mLang, langPrefix)) {
*aRetval = voice;
return true;
}
}
for (int32_t i = mVoices.Length(); i > 0; ) {
VoiceData* voice = mVoices[--i];
if (StringBeginsWith(voice->mLang, langPrefix)) {
*aRetval = voice;
return true;
}
}
dashPos = end;
end = start;
if (!RFindInReadable(NS_LITERAL_STRING("-"), end, dashPos)) {
break;
}
}
return false;
}
VoiceData*
nsSynthVoiceRegistry::FindBestMatch(const nsAString& aUri,
const nsAString& aLang)
{
if (mVoices.IsEmpty()) {
return nullptr;
}
bool found = false;
VoiceData* retval = mUriVoiceMap.GetWeak(aUri, &found);
if (found) {
LOG(LogLevel::Debug, ("nsSynthVoiceRegistry::FindBestMatch - Matched URI"));
return retval;
}
// Try finding a match for given voice.
if (!aLang.IsVoid() && !aLang.IsEmpty()) {
if (FindVoiceByLang(aLang, &retval)) {
LOG(LogLevel::Debug,
("nsSynthVoiceRegistry::FindBestMatch - Matched language (%s ~= %s)",
NS_ConvertUTF16toUTF8(aLang).get(),
NS_ConvertUTF16toUTF8(retval->mLang).get()));
return retval;
}
}
// Try UI language.
nsresult rv;
nsCOMPtr<nsILocaleService> localeService = do_GetService(NS_LOCALESERVICE_CONTRACTID, &rv);
if (NS_WARN_IF(NS_FAILED(rv))) {
return nullptr;
}
nsAutoString uiLang;
rv = localeService->GetLocaleComponentForUserAgent(uiLang);
if (NS_WARN_IF(NS_FAILED(rv))) {
return nullptr;
}
if (FindVoiceByLang(uiLang, &retval)) {
LOG(LogLevel::Debug,
("nsSynthVoiceRegistry::FindBestMatch - Matched UI language (%s ~= %s)",
NS_ConvertUTF16toUTF8(uiLang).get(),
NS_ConvertUTF16toUTF8(retval->mLang).get()));
return retval;
}
// Try en-US, the language of locale "C"
if (FindVoiceByLang(NS_LITERAL_STRING("en-US"), &retval)) {
LOG(LogLevel::Debug,
("nsSynthVoiceRegistry::FindBestMatch - Matched C locale language (en-US ~= %s)",
NS_ConvertUTF16toUTF8(retval->mLang).get()));
return retval;
}
// The top default voice is better than nothing...
if (!mDefaultVoices.IsEmpty()) {
return mDefaultVoices.LastElement();
}
return nullptr;
}
already_AddRefed<nsSpeechTask>
nsSynthVoiceRegistry::SpeakUtterance(SpeechSynthesisUtterance& aUtterance,
const nsAString& aDocLang)
{
nsString lang = nsString(aUtterance.mLang.IsEmpty() ? aDocLang : aUtterance.mLang);
nsAutoString uri;
if (aUtterance.mVoice) {
aUtterance.mVoice->GetVoiceURI(uri);
}
// Get current audio volume to apply speech call
float volume = aUtterance.Volume();
RefPtr<AudioChannelService> service = AudioChannelService::GetOrCreate();
if (service) {
if (nsCOMPtr<nsPIDOMWindowInner> topWindow = aUtterance.GetOwner()) {
// TODO : use audio channel agent, open new bug to fix it.
uint32_t channel = static_cast<uint32_t>(AudioChannelService::GetDefaultAudioChannel());
AudioPlaybackConfig config = service->GetMediaConfig(topWindow->GetOuterWindow(),
channel);
volume = config.mMuted ? 0.0f : config.mVolume * volume;
}
}
RefPtr<nsSpeechTask> task;
if (XRE_IsContentProcess()) {
task = new SpeechTaskChild(&aUtterance);
SpeechSynthesisRequestChild* actor =
new SpeechSynthesisRequestChild(static_cast<SpeechTaskChild*>(task.get()));
mSpeechSynthChild->SendPSpeechSynthesisRequestConstructor(actor,
aUtterance.mText,
lang,
uri,
volume,
aUtterance.Rate(),
aUtterance.Pitch());
} else {
task = new nsSpeechTask(&aUtterance);
Speak(aUtterance.mText, lang, uri,
volume, aUtterance.Rate(), aUtterance.Pitch(), task);
}
return task.forget();
}
void
nsSynthVoiceRegistry::Speak(const nsAString& aText,
const nsAString& aLang,
const nsAString& aUri,
const float& aVolume,
const float& aRate,
const float& aPitch,
nsSpeechTask* aTask)
{
MOZ_ASSERT(XRE_IsParentProcess());
VoiceData* voice = FindBestMatch(aUri, aLang);
if (!voice) {
NS_WARNING("No voices found.");
aTask->DispatchError(0, 0);
return;
}
aTask->SetChosenVoiceURI(voice->mUri);
if (mUseGlobalQueue || MediaPrefs::WebSpeechForceGlobal()) {
LOG(LogLevel::Debug,
("nsSynthVoiceRegistry::Speak queueing text='%s' lang='%s' uri='%s' rate=%f pitch=%f",
NS_ConvertUTF16toUTF8(aText).get(), NS_ConvertUTF16toUTF8(aLang).get(),
NS_ConvertUTF16toUTF8(aUri).get(), aRate, aPitch));
RefPtr<GlobalQueueItem> item = new GlobalQueueItem(voice, aTask, aText,
aVolume, aRate, aPitch);
mGlobalQueue.AppendElement(item);
if (mGlobalQueue.Length() == 1) {
SpeakImpl(item->mVoice, item->mTask, item->mText, item->mVolume, item->mRate,
item->mPitch);
}
} else {
SpeakImpl(voice, aTask, aText, aVolume, aRate, aPitch);
}
}
void
nsSynthVoiceRegistry::SpeakNext()
{
MOZ_ASSERT(XRE_IsParentProcess());
LOG(LogLevel::Debug,
("nsSynthVoiceRegistry::SpeakNext %d", mGlobalQueue.IsEmpty()));
SetIsSpeaking(false);
if (mGlobalQueue.IsEmpty()) {
return;
}
mGlobalQueue.RemoveElementAt(0);
while (!mGlobalQueue.IsEmpty()) {
RefPtr<GlobalQueueItem> item = mGlobalQueue.ElementAt(0);
if (item->mTask->IsPreCanceled()) {
mGlobalQueue.RemoveElementAt(0);
continue;
}
if (!item->mTask->IsPrePaused()) {
SpeakImpl(item->mVoice, item->mTask, item->mText, item->mVolume,
item->mRate, item->mPitch);
}
break;
}
}
void
nsSynthVoiceRegistry::ResumeQueue()
{
MOZ_ASSERT(XRE_IsParentProcess());
LOG(LogLevel::Debug,
("nsSynthVoiceRegistry::ResumeQueue %d", mGlobalQueue.IsEmpty()));
if (mGlobalQueue.IsEmpty()) {
return;
}
RefPtr<GlobalQueueItem> item = mGlobalQueue.ElementAt(0);
if (!item->mTask->IsPrePaused()) {
SpeakImpl(item->mVoice, item->mTask, item->mText, item->mVolume,
item->mRate, item->mPitch);
}
}
bool
nsSynthVoiceRegistry::IsSpeaking()
{
return mIsSpeaking;
}
void
nsSynthVoiceRegistry::SetIsSpeaking(bool aIsSpeaking)
{
MOZ_ASSERT(XRE_IsParentProcess());
// Only set to 'true' if global queue is enabled.
mIsSpeaking =
aIsSpeaking && (mUseGlobalQueue || MediaPrefs::WebSpeechForceGlobal());
nsTArray<SpeechSynthesisParent*> ssplist;
GetAllSpeechSynthActors(ssplist);
for (uint32_t i = 0; i < ssplist.Length(); ++i) {
Unused << ssplist[i]->SendIsSpeakingChanged(aIsSpeaking);
}
}
void
nsSynthVoiceRegistry::SpeakImpl(VoiceData* aVoice,
nsSpeechTask* aTask,
const nsAString& aText,
const float& aVolume,
const float& aRate,
const float& aPitch)
{
LOG(LogLevel::Debug,
("nsSynthVoiceRegistry::SpeakImpl queueing text='%s' uri='%s' rate=%f pitch=%f",
NS_ConvertUTF16toUTF8(aText).get(), NS_ConvertUTF16toUTF8(aVoice->mUri).get(),
aRate, aPitch));
SpeechServiceType serviceType;
DebugOnly<nsresult> rv = aVoice->mService->GetServiceType(&serviceType);
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv), "Failed to get speech service type");
if (serviceType == nsISpeechService::SERVICETYPE_INDIRECT_AUDIO) {
aTask->InitIndirectAudio();
} else {
aTask->InitDirectAudio();
}
if (NS_FAILED(aVoice->mService->Speak(aText, aVoice->mUri, aVolume, aRate,
aPitch, aTask))) {
if (serviceType == nsISpeechService::SERVICETYPE_INDIRECT_AUDIO) {
aTask->DispatchError(0, 0);
}
// XXX When using direct audio, no way to dispatch error
}
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,109 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_nsSynthVoiceRegistry_h
#define mozilla_dom_nsSynthVoiceRegistry_h
#include "nsISynthVoiceRegistry.h"
#include "nsRefPtrHashtable.h"
#include "nsTArray.h"
#include "MediaStreamGraph.h"
class nsISpeechService;
namespace mozilla {
namespace dom {
class RemoteVoice;
class SpeechSynthesisUtterance;
class SpeechSynthesisChild;
class nsSpeechTask;
class VoiceData;
class GlobalQueueItem;
class nsSynthVoiceRegistry final : public nsISynthVoiceRegistry
{
public:
NS_DECL_ISUPPORTS
NS_DECL_NSISYNTHVOICEREGISTRY
nsSynthVoiceRegistry();
already_AddRefed<nsSpeechTask> SpeakUtterance(SpeechSynthesisUtterance& aUtterance,
const nsAString& aDocLang);
void Speak(const nsAString& aText, const nsAString& aLang,
const nsAString& aUri, const float& aVolume, const float& aRate,
const float& aPitch, nsSpeechTask* aTask);
void SendVoicesAndState(InfallibleTArray<RemoteVoice>* aVoices,
InfallibleTArray<nsString>* aDefaults,
bool* aIsSpeaking);
void SpeakNext();
void ResumeQueue();
bool IsSpeaking();
void SetIsSpeaking(bool aIsSpeaking);
static nsSynthVoiceRegistry* GetInstance();
static already_AddRefed<nsSynthVoiceRegistry> GetInstanceForService();
static void RecvRemoveVoice(const nsAString& aUri);
static void RecvAddVoice(const RemoteVoice& aVoice);
static void RecvSetDefaultVoice(const nsAString& aUri, bool aIsDefault);
static void RecvIsSpeakingChanged(bool aIsSpeaking);
static void RecvNotifyVoicesChanged();
static void Shutdown();
private:
virtual ~nsSynthVoiceRegistry();
VoiceData* FindBestMatch(const nsAString& aUri, const nsAString& lang);
bool FindVoiceByLang(const nsAString& aLang, VoiceData** aRetval);
nsresult AddVoiceImpl(nsISpeechService* aService,
const nsAString& aUri,
const nsAString& aName,
const nsAString& aLang,
bool aLocalService,
bool aQueuesUtterances);
void SpeakImpl(VoiceData* aVoice,
nsSpeechTask* aTask,
const nsAString& aText,
const float& aVolume,
const float& aRate,
const float& aPitch);
nsTArray<RefPtr<VoiceData>> mVoices;
nsTArray<RefPtr<VoiceData>> mDefaultVoices;
nsRefPtrHashtable<nsStringHashKey, VoiceData> mUriVoiceMap;
SpeechSynthesisChild* mSpeechSynthChild;
bool mUseGlobalQueue;
nsTArray<RefPtr<GlobalQueueItem>> mGlobalQueue;
bool mIsSpeaking;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,58 @@
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "mozilla/ModuleUtils.h"
#include "nsIClassInfoImpl.h"
#ifdef MOZ_WEBRTC
#include "nsPicoService.h"
using namespace mozilla::dom;
#define PICOSERVICE_CID \
{0x346c4fc8, 0x12fe, 0x459c, {0x81, 0x19, 0x9a, 0xa7, 0x73, 0x37, 0x7f, 0xf4}}
#define PICOSERVICE_CONTRACTID "@mozilla.org/synthpico;1"
// Defines nsPicoServiceConstructor
NS_GENERIC_FACTORY_SINGLETON_CONSTRUCTOR(nsPicoService,
nsPicoService::GetInstanceForService)
// Defines kPICOSERVICE_CID
NS_DEFINE_NAMED_CID(PICOSERVICE_CID);
static const mozilla::Module::CIDEntry kCIDs[] = {
{ &kPICOSERVICE_CID, true, nullptr, nsPicoServiceConstructor },
{ nullptr }
};
static const mozilla::Module::ContractIDEntry kContracts[] = {
{ PICOSERVICE_CONTRACTID, &kPICOSERVICE_CID },
{ nullptr }
};
static const mozilla::Module::CategoryEntry kCategories[] = {
{ "profile-after-change", "Pico Speech Synth", PICOSERVICE_CONTRACTID },
{ nullptr }
};
static void
UnloadPicoModule()
{
nsPicoService::Shutdown();
}
static const mozilla::Module kModule = {
mozilla::Module::kVersion,
kCIDs,
kContracts,
kCategories,
nullptr,
nullptr,
UnloadPicoModule
};
NSMODULE_DEFN(synthpico) = &kModule;
#endif

View file

@ -0,0 +1,13 @@
# -*- Mode: python; indent-tabs-mode: nil; tab-width: 40 -*-
# vim: set filetype=python:
# This Source Code Form is subject to the terms of the Mozilla Public
# License, v. 2.0. If a copy of the MPL was not distributed with this
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
UNIFIED_SOURCES += [
'nsPicoService.cpp',
'PicoModule.cpp'
]
include('/ipc/chromium/chromium-config.mozbuild')
FINAL_LIBRARY = 'xul'

View file

@ -0,0 +1,761 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsISupports.h"
#include "nsPicoService.h"
#include "nsPrintfCString.h"
#include "nsIWeakReferenceUtils.h"
#include "SharedBuffer.h"
#include "nsISimpleEnumerator.h"
#include "mozilla/dom/nsSynthVoiceRegistry.h"
#include "mozilla/dom/nsSpeechTask.h"
#include "nsIFile.h"
#include "nsThreadUtils.h"
#include "prenv.h"
#include "mozilla/Preferences.h"
#include "mozilla/DebugOnly.h"
#include <dlfcn.h>
// Pico API constants
// Size of memory allocated for pico engine and voice resources.
// We only have one voice and its resources loaded at once, so this
// should always be enough.
#define PICO_MEM_SIZE 2500000
// Max length of returned strings. Pico will never return longer strings,
// so this amount should be good enough for preallocating.
#define PICO_RETSTRINGSIZE 200
// Max amount we want from a single call of pico_getData
#define PICO_MAX_CHUNK_SIZE 128
// Arbitrary name for loaded voice, it doesn't mean anything outside of Pico
#define PICO_VOICE_NAME "pico"
// Return status from pico_getData meaning there is more data in the pipeline
// to get from more calls to pico_getData
#define PICO_STEP_BUSY 201
// For performing a "soft" reset between utterances. This is used when one
// utterance is interrupted by a new one.
#define PICO_RESET_SOFT 0x10
// Currently, Pico only provides mono output.
#define PICO_CHANNELS_NUM 1
// Pico's sample rate is always 16000
#define PICO_SAMPLE_RATE 16000
// The path to the language files in Gonk
#define GONK_PICO_LANG_PATH "/system/tts/lang_pico"
namespace mozilla {
namespace dom {
StaticRefPtr<nsPicoService> nsPicoService::sSingleton;
class PicoApi
{
public:
PicoApi() : mInitialized(false) {}
bool Init()
{
if (mInitialized) {
return true;
}
void* handle = dlopen("libttspico.so", RTLD_LAZY);
if (!handle) {
NS_WARNING("Failed to open libttspico.so, pico cannot run");
return false;
}
pico_initialize =
(pico_Status (*)(void*, uint32_t, pico_System*))dlsym(
handle, "pico_initialize");
pico_terminate =
(pico_Status (*)(pico_System*))dlsym(handle, "pico_terminate");
pico_getSystemStatusMessage =
(pico_Status (*)(pico_System, pico_Status, pico_Retstring))dlsym(
handle, "pico_getSystemStatusMessage");;
pico_loadResource =
(pico_Status (*)(pico_System, const char*, pico_Resource*))dlsym(
handle, "pico_loadResource");
pico_unloadResource =
(pico_Status (*)(pico_System, pico_Resource*))dlsym(
handle, "pico_unloadResource");
pico_getResourceName =
(pico_Status (*)(pico_System, pico_Resource, pico_Retstring))dlsym(
handle, "pico_getResourceName");
pico_createVoiceDefinition =
(pico_Status (*)(pico_System, const char*))dlsym(
handle, "pico_createVoiceDefinition");
pico_addResourceToVoiceDefinition =
(pico_Status (*)(pico_System, const char*, const char*))dlsym(
handle, "pico_addResourceToVoiceDefinition");
pico_releaseVoiceDefinition =
(pico_Status (*)(pico_System, const char*))dlsym(
handle, "pico_releaseVoiceDefinition");
pico_newEngine =
(pico_Status (*)(pico_System, const char*, pico_Engine*))dlsym(
handle, "pico_newEngine");
pico_disposeEngine =
(pico_Status (*)(pico_System, pico_Engine*))dlsym(
handle, "pico_disposeEngine");
pico_resetEngine =
(pico_Status (*)(pico_Engine, int32_t))dlsym(handle, "pico_resetEngine");
pico_putTextUtf8 =
(pico_Status (*)(pico_Engine, const char*, const int16_t, int16_t*))dlsym(
handle, "pico_putTextUtf8");
pico_getData =
(pico_Status (*)(pico_Engine, void*, int16_t, int16_t*, int16_t*))dlsym(
handle, "pico_getData");
mInitialized = true;
return true;
}
typedef signed int pico_Status;
typedef char pico_Retstring[PICO_RETSTRINGSIZE];
pico_Status (* pico_initialize)(void*, uint32_t, pico_System*);
pico_Status (* pico_terminate)(pico_System*);
pico_Status (* pico_getSystemStatusMessage)(
pico_System, pico_Status, pico_Retstring);
pico_Status (* pico_loadResource)(pico_System, const char*, pico_Resource*);
pico_Status (* pico_unloadResource)(pico_System, pico_Resource*);
pico_Status (* pico_getResourceName)(
pico_System, pico_Resource, pico_Retstring);
pico_Status (* pico_createVoiceDefinition)(pico_System, const char*);
pico_Status (* pico_addResourceToVoiceDefinition)(
pico_System, const char*, const char*);
pico_Status (* pico_releaseVoiceDefinition)(pico_System, const char*);
pico_Status (* pico_newEngine)(pico_System, const char*, pico_Engine*);
pico_Status (* pico_disposeEngine)(pico_System, pico_Engine*);
pico_Status (* pico_resetEngine)(pico_Engine, int32_t);
pico_Status (* pico_putTextUtf8)(
pico_Engine, const char*, const int16_t, int16_t*);
pico_Status (* pico_getData)(
pico_Engine, void*, const int16_t, int16_t*, int16_t*);
private:
bool mInitialized;
} sPicoApi;
#define PICO_ENSURE_SUCCESS_VOID(_funcName, _status) \
if (_status < 0) { \
PicoApi::pico_Retstring message; \
sPicoApi.pico_getSystemStatusMessage( \
nsPicoService::sSingleton->mPicoSystem, _status, message); \
NS_WARNING( \
nsPrintfCString("Error running %s: %s", _funcName, message).get()); \
return; \
}
#define PICO_ENSURE_SUCCESS(_funcName, _status, _rv) \
if (_status < 0) { \
PicoApi::pico_Retstring message; \
sPicoApi.pico_getSystemStatusMessage( \
nsPicoService::sSingleton->mPicoSystem, _status, message); \
NS_WARNING( \
nsPrintfCString("Error running %s: %s", _funcName, message).get()); \
return _rv; \
}
class PicoVoice
{
public:
PicoVoice(const nsAString& aLanguage)
: mLanguage(aLanguage) {}
NS_INLINE_DECL_THREADSAFE_REFCOUNTING(PicoVoice)
// Voice language, in BCB-47 syntax
nsString mLanguage;
// Language resource file
nsCString mTaFile;
// Speaker resource file
nsCString mSgFile;
private:
~PicoVoice() {}
};
class PicoCallbackRunnable : public Runnable,
public nsISpeechTaskCallback
{
friend class PicoSynthDataRunnable;
public:
PicoCallbackRunnable(const nsAString& aText, PicoVoice* aVoice,
float aRate, float aPitch, nsISpeechTask* aTask,
nsPicoService* aService)
: mText(NS_ConvertUTF16toUTF8(aText))
, mRate(aRate)
, mPitch(aPitch)
, mFirstData(true)
, mTask(aTask)
, mVoice(aVoice)
, mService(aService) { }
NS_DECL_ISUPPORTS_INHERITED
NS_DECL_NSISPEECHTASKCALLBACK
NS_IMETHOD Run() override;
bool IsCurrentTask() { return mService->mCurrentTask == mTask; }
private:
~PicoCallbackRunnable() { }
void DispatchSynthDataRunnable(already_AddRefed<SharedBuffer>&& aBuffer,
size_t aBufferSize);
nsCString mText;
float mRate;
float mPitch;
bool mFirstData;
// We use this pointer to compare it with the current service task.
// If they differ, this runnable should stop.
nsISpeechTask* mTask;
// We hold a strong reference to the service, which in turn holds
// a strong reference to this voice.
PicoVoice* mVoice;
// By holding a strong reference to the service we guarantee that it won't be
// destroyed before this runnable.
RefPtr<nsPicoService> mService;
};
NS_IMPL_ISUPPORTS_INHERITED(PicoCallbackRunnable, Runnable, nsISpeechTaskCallback)
// Runnable
NS_IMETHODIMP
PicoCallbackRunnable::Run()
{
MOZ_ASSERT(!NS_IsMainThread());
PicoApi::pico_Status status = 0;
if (mService->CurrentVoice() != mVoice) {
mService->LoadEngine(mVoice);
} else {
status = sPicoApi.pico_resetEngine(mService->mPicoEngine, PICO_RESET_SOFT);
PICO_ENSURE_SUCCESS("pico_unloadResource", status, NS_ERROR_FAILURE);
}
// Add SSML markup for pitch and rate. Pico uses a minimal parser,
// so no namespace is needed.
nsPrintfCString markedUpText(
"<pitch level=\"%0.0f\"><speed level=\"%0.0f\">%s</speed></pitch>",
std::min(std::max(50.0f, mPitch * 100), 200.0f),
std::min(std::max(20.0f, mRate * 100), 500.0f),
mText.get());
const char* text = markedUpText.get();
size_t buffer_size = 512, buffer_offset = 0;
RefPtr<SharedBuffer> buffer = SharedBuffer::Create(buffer_size);
int16_t text_offset = 0, bytes_recv = 0, bytes_sent = 0, out_data_type = 0;
int16_t text_remaining = markedUpText.Length() + 1;
// Run this loop while this is the current task
while (IsCurrentTask()) {
if (text_remaining) {
status = sPicoApi.pico_putTextUtf8(mService->mPicoEngine,
text + text_offset, text_remaining,
&bytes_sent);
PICO_ENSURE_SUCCESS("pico_putTextUtf8", status, NS_ERROR_FAILURE);
// XXX: End speech task on error
text_remaining -= bytes_sent;
text_offset += bytes_sent;
} else {
// If we already fed all the text to the engine, send a zero length buffer
// and quit.
DispatchSynthDataRunnable(already_AddRefed<SharedBuffer>(), 0);
break;
}
do {
// Run this loop while the result of getData is STEP_BUSY, when it finishes
// synthesizing audio for the given text, it returns STEP_IDLE. We then
// break to the outer loop and feed more text, if there is any left.
if (!IsCurrentTask()) {
// If the task has changed, quit.
break;
}
if (buffer_size - buffer_offset < PICO_MAX_CHUNK_SIZE) {
// The next audio chunk retrieved may be bigger than our buffer,
// so send the data and flush the buffer.
DispatchSynthDataRunnable(buffer.forget(), buffer_offset);
buffer_offset = 0;
buffer = SharedBuffer::Create(buffer_size);
}
status = sPicoApi.pico_getData(mService->mPicoEngine,
(uint8_t*)buffer->Data() + buffer_offset,
PICO_MAX_CHUNK_SIZE,
&bytes_recv, &out_data_type);
PICO_ENSURE_SUCCESS("pico_getData", status, NS_ERROR_FAILURE);
buffer_offset += bytes_recv;
} while (status == PICO_STEP_BUSY);
}
return NS_OK;
}
void
PicoCallbackRunnable::DispatchSynthDataRunnable(
already_AddRefed<SharedBuffer>&& aBuffer, size_t aBufferSize)
{
class PicoSynthDataRunnable final : public Runnable
{
public:
PicoSynthDataRunnable(already_AddRefed<SharedBuffer>& aBuffer,
size_t aBufferSize, bool aFirstData,
PicoCallbackRunnable* aCallback)
: mBuffer(aBuffer)
, mBufferSize(aBufferSize)
, mFirstData(aFirstData)
, mCallback(aCallback) {
}
NS_IMETHOD Run() override
{
MOZ_ASSERT(NS_IsMainThread());
if (!mCallback->IsCurrentTask()) {
return NS_ERROR_NOT_AVAILABLE;
}
nsISpeechTask* task = mCallback->mTask;
if (mFirstData) {
task->Setup(mCallback, PICO_CHANNELS_NUM, PICO_SAMPLE_RATE, 2);
}
return task->SendAudioNative(
mBufferSize ? static_cast<short*>(mBuffer->Data()) : nullptr, mBufferSize / 2);
}
private:
RefPtr<SharedBuffer> mBuffer;
size_t mBufferSize;
bool mFirstData;
RefPtr<PicoCallbackRunnable> mCallback;
};
nsCOMPtr<nsIRunnable> sendEvent =
new PicoSynthDataRunnable(aBuffer, aBufferSize, mFirstData, this);
NS_DispatchToMainThread(sendEvent);
mFirstData = false;
}
// nsISpeechTaskCallback
NS_IMETHODIMP
PicoCallbackRunnable::OnPause()
{
return NS_OK;
}
NS_IMETHODIMP
PicoCallbackRunnable::OnResume()
{
return NS_OK;
}
NS_IMETHODIMP
PicoCallbackRunnable::OnCancel()
{
mService->mCurrentTask = nullptr;
return NS_OK;
}
NS_IMETHODIMP
PicoCallbackRunnable::OnVolumeChanged(float aVolume)
{
return NS_OK;
}
NS_INTERFACE_MAP_BEGIN(nsPicoService)
NS_INTERFACE_MAP_ENTRY(nsISpeechService)
NS_INTERFACE_MAP_ENTRY(nsIObserver)
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsIObserver)
NS_INTERFACE_MAP_END
NS_IMPL_ADDREF(nsPicoService)
NS_IMPL_RELEASE(nsPicoService)
nsPicoService::nsPicoService()
: mInitialized(false)
, mVoicesMonitor("nsPicoService::mVoices")
, mCurrentTask(nullptr)
, mPicoSystem(nullptr)
, mPicoEngine(nullptr)
, mSgResource(nullptr)
, mTaResource(nullptr)
, mPicoMemArea(nullptr)
{
}
nsPicoService::~nsPicoService()
{
// We don't worry about removing the voices because this gets
// destructed at shutdown along with the voice registry.
MonitorAutoLock autoLock(mVoicesMonitor);
mVoices.Clear();
if (mThread) {
mThread->Shutdown();
}
UnloadEngine();
}
// nsIObserver
NS_IMETHODIMP
nsPicoService::Observe(nsISupports* aSubject, const char* aTopic,
const char16_t* aData)
{
MOZ_ASSERT(NS_IsMainThread());
if(NS_WARN_IF(!(!strcmp(aTopic, "profile-after-change")))) {
return NS_ERROR_UNEXPECTED;
}
if (!Preferences::GetBool("media.webspeech.synth.enabled") ||
Preferences::GetBool("media.webspeech.synth.test")) {
return NS_OK;
}
DebugOnly<nsresult> rv = NS_NewNamedThread("Pico Worker", getter_AddRefs(mThread));
MOZ_ASSERT(NS_SUCCEEDED(rv));
return mThread->Dispatch(
NewRunnableMethod(this, &nsPicoService::Init), NS_DISPATCH_NORMAL);
}
// nsISpeechService
NS_IMETHODIMP
nsPicoService::Speak(const nsAString& aText, const nsAString& aUri,
float aVolume, float aRate, float aPitch,
nsISpeechTask* aTask)
{
if(NS_WARN_IF(!(mInitialized))) {
return NS_ERROR_NOT_AVAILABLE;
}
MonitorAutoLock autoLock(mVoicesMonitor);
bool found = false;
PicoVoice* voice = mVoices.GetWeak(aUri, &found);
if(NS_WARN_IF(!(found))) {
return NS_ERROR_NOT_AVAILABLE;
}
mCurrentTask = aTask;
RefPtr<PicoCallbackRunnable> cb = new PicoCallbackRunnable(aText, voice, aRate, aPitch, aTask, this);
return mThread->Dispatch(cb, NS_DISPATCH_NORMAL);
}
NS_IMETHODIMP
nsPicoService::GetServiceType(SpeechServiceType* aServiceType)
{
*aServiceType = nsISpeechService::SERVICETYPE_DIRECT_AUDIO;
return NS_OK;
}
// private methods
void
nsPicoService::Init()
{
MOZ_ASSERT(!NS_IsMainThread());
MOZ_ASSERT(!mInitialized);
if (!sPicoApi.Init()) {
NS_WARNING("Failed to initialize pico library");
return;
}
// Use environment variable, or default android/b2g path
nsAutoCString langPath(PR_GetEnv("PICO_LANG_PATH"));
if (langPath.IsEmpty()) {
langPath.AssignLiteral(GONK_PICO_LANG_PATH);
}
nsCOMPtr<nsIFile> voicesDir;
NS_NewNativeLocalFile(langPath, true, getter_AddRefs(voicesDir));
nsCOMPtr<nsISimpleEnumerator> dirIterator;
nsresult rv = voicesDir->GetDirectoryEntries(getter_AddRefs(dirIterator));
if (NS_FAILED(rv)) {
NS_WARNING(nsPrintfCString("Failed to get contents of directory: %s", langPath.get()).get());
return;
}
bool hasMoreElements = false;
rv = dirIterator->HasMoreElements(&hasMoreElements);
MOZ_ASSERT(NS_SUCCEEDED(rv));
MonitorAutoLock autoLock(mVoicesMonitor);
while (hasMoreElements && NS_SUCCEEDED(rv)) {
nsCOMPtr<nsISupports> supports;
rv = dirIterator->GetNext(getter_AddRefs(supports));
MOZ_ASSERT(NS_SUCCEEDED(rv));
nsCOMPtr<nsIFile> voiceFile = do_QueryInterface(supports);
MOZ_ASSERT(voiceFile);
nsAutoCString leafName;
voiceFile->GetNativeLeafName(leafName);
nsAutoString lang;
if (GetVoiceFileLanguage(leafName, lang)) {
nsAutoString uri;
uri.AssignLiteral("urn:moz-tts:pico:");
uri.Append(lang);
bool found = false;
PicoVoice* voice = mVoices.GetWeak(uri, &found);
if (!found) {
voice = new PicoVoice(lang);
mVoices.Put(uri, voice);
}
// Each voice consists of two lingware files: A language resource file,
// suffixed by _ta.bin, and a speaker resource file, suffixed by _sb.bin.
// We currently assume that there is a pair of files for each language.
if (StringEndsWith(leafName, NS_LITERAL_CSTRING("_ta.bin"))) {
rv = voiceFile->GetPersistentDescriptor(voice->mTaFile);
MOZ_ASSERT(NS_SUCCEEDED(rv));
} else if (StringEndsWith(leafName, NS_LITERAL_CSTRING("_sg.bin"))) {
rv = voiceFile->GetPersistentDescriptor(voice->mSgFile);
MOZ_ASSERT(NS_SUCCEEDED(rv));
}
}
rv = dirIterator->HasMoreElements(&hasMoreElements);
}
NS_DispatchToMainThread(NewRunnableMethod(this, &nsPicoService::RegisterVoices));
}
void
nsPicoService::RegisterVoices()
{
nsSynthVoiceRegistry* registry = nsSynthVoiceRegistry::GetInstance();
for (auto iter = mVoices.Iter(); !iter.Done(); iter.Next()) {
const nsAString& uri = iter.Key();
RefPtr<PicoVoice>& voice = iter.Data();
// If we are missing either a language or a voice resource, it is invalid.
if (voice->mTaFile.IsEmpty() || voice->mSgFile.IsEmpty()) {
iter.Remove();
continue;
}
nsAutoString name;
name.AssignLiteral("Pico ");
name.Append(voice->mLanguage);
// This service is multi-threaded and can handle more than one utterance at a
// time before previous utterances end. So, aQueuesUtterances == false
DebugOnly<nsresult> rv =
registry->AddVoice(this, uri, name, voice->mLanguage, true, false);
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv), "Failed to add voice");
}
mInitialized = true;
}
bool
nsPicoService::GetVoiceFileLanguage(const nsACString& aFileName, nsAString& aLang)
{
nsACString::const_iterator start, end;
aFileName.BeginReading(start);
aFileName.EndReading(end);
// The lingware filename syntax is language_(ta/sg).bin,
// we extract the language prefix here.
if (FindInReadable(NS_LITERAL_CSTRING("_"), start, end)) {
end = start;
aFileName.BeginReading(start);
aLang.Assign(NS_ConvertUTF8toUTF16(Substring(start, end)));
return true;
}
return false;
}
void
nsPicoService::LoadEngine(PicoVoice* aVoice)
{
PicoApi::pico_Status status = 0;
if (mPicoSystem) {
UnloadEngine();
}
if (!mPicoMemArea) {
mPicoMemArea = MakeUnique<uint8_t[]>(PICO_MEM_SIZE);
}
status = sPicoApi.pico_initialize(mPicoMemArea.get(),
PICO_MEM_SIZE, &mPicoSystem);
PICO_ENSURE_SUCCESS_VOID("pico_initialize", status);
status = sPicoApi.pico_loadResource(mPicoSystem, aVoice->mTaFile.get(), &mTaResource);
PICO_ENSURE_SUCCESS_VOID("pico_loadResource", status);
status = sPicoApi.pico_loadResource(mPicoSystem, aVoice->mSgFile.get(), &mSgResource);
PICO_ENSURE_SUCCESS_VOID("pico_loadResource", status);
status = sPicoApi.pico_createVoiceDefinition(mPicoSystem, PICO_VOICE_NAME);
PICO_ENSURE_SUCCESS_VOID("pico_createVoiceDefinition", status);
char taName[PICO_RETSTRINGSIZE];
status = sPicoApi.pico_getResourceName(mPicoSystem, mTaResource, taName);
PICO_ENSURE_SUCCESS_VOID("pico_getResourceName", status);
status = sPicoApi.pico_addResourceToVoiceDefinition(
mPicoSystem, PICO_VOICE_NAME, taName);
PICO_ENSURE_SUCCESS_VOID("pico_addResourceToVoiceDefinition", status);
char sgName[PICO_RETSTRINGSIZE];
status = sPicoApi.pico_getResourceName(mPicoSystem, mSgResource, sgName);
PICO_ENSURE_SUCCESS_VOID("pico_getResourceName", status);
status = sPicoApi.pico_addResourceToVoiceDefinition(
mPicoSystem, PICO_VOICE_NAME, sgName);
PICO_ENSURE_SUCCESS_VOID("pico_addResourceToVoiceDefinition", status);
status = sPicoApi.pico_newEngine(mPicoSystem, PICO_VOICE_NAME, &mPicoEngine);
PICO_ENSURE_SUCCESS_VOID("pico_newEngine", status);
if (sSingleton) {
sSingleton->mCurrentVoice = aVoice;
}
}
void
nsPicoService::UnloadEngine()
{
PicoApi::pico_Status status = 0;
if (mPicoEngine) {
status = sPicoApi.pico_disposeEngine(mPicoSystem, &mPicoEngine);
PICO_ENSURE_SUCCESS_VOID("pico_disposeEngine", status);
status = sPicoApi.pico_releaseVoiceDefinition(mPicoSystem, PICO_VOICE_NAME);
PICO_ENSURE_SUCCESS_VOID("pico_releaseVoiceDefinition", status);
mPicoEngine = nullptr;
}
if (mSgResource) {
status = sPicoApi.pico_unloadResource(mPicoSystem, &mSgResource);
PICO_ENSURE_SUCCESS_VOID("pico_unloadResource", status);
mSgResource = nullptr;
}
if (mTaResource) {
status = sPicoApi.pico_unloadResource(mPicoSystem, &mTaResource);
PICO_ENSURE_SUCCESS_VOID("pico_unloadResource", status);
mTaResource = nullptr;
}
if (mPicoSystem) {
status = sPicoApi.pico_terminate(&mPicoSystem);
PICO_ENSURE_SUCCESS_VOID("pico_terminate", status);
mPicoSystem = nullptr;
}
}
PicoVoice*
nsPicoService::CurrentVoice()
{
MOZ_ASSERT(!NS_IsMainThread());
return mCurrentVoice;
}
// static methods
nsPicoService*
nsPicoService::GetInstance()
{
MOZ_ASSERT(NS_IsMainThread());
if (!XRE_IsParentProcess()) {
MOZ_ASSERT(false, "nsPicoService can only be started on main gecko process");
return nullptr;
}
if (!sSingleton) {
sSingleton = new nsPicoService();
}
return sSingleton;
}
already_AddRefed<nsPicoService>
nsPicoService::GetInstanceForService()
{
RefPtr<nsPicoService> picoService = GetInstance();
return picoService.forget();
}
void
nsPicoService::Shutdown()
{
if (!sSingleton) {
return;
}
sSingleton->mCurrentTask = nullptr;
sSingleton = nullptr;
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,93 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef nsPicoService_h
#define nsPicoService_h
#include "mozilla/Mutex.h"
#include "nsTArray.h"
#include "nsIObserver.h"
#include "nsIThread.h"
#include "nsISpeechService.h"
#include "nsRefPtrHashtable.h"
#include "mozilla/StaticPtr.h"
#include "mozilla/Monitor.h"
#include "mozilla/UniquePtr.h"
namespace mozilla {
namespace dom {
class PicoVoice;
class PicoCallbackRunnable;
typedef void* pico_System;
typedef void* pico_Resource;
typedef void* pico_Engine;
class nsPicoService : public nsIObserver,
public nsISpeechService
{
friend class PicoCallbackRunnable;
friend class PicoInitRunnable;
public:
NS_DECL_THREADSAFE_ISUPPORTS
NS_DECL_NSISPEECHSERVICE
NS_DECL_NSIOBSERVER
nsPicoService();
static nsPicoService* GetInstance();
static already_AddRefed<nsPicoService> GetInstanceForService();
static void Shutdown();
private:
virtual ~nsPicoService();
void Init();
void RegisterVoices();
bool GetVoiceFileLanguage(const nsACString& aFileName, nsAString& aLang);
void LoadEngine(PicoVoice* aVoice);
void UnloadEngine();
PicoVoice* CurrentVoice();
bool mInitialized;
nsCOMPtr<nsIThread> mThread;
nsRefPtrHashtable<nsStringHashKey, PicoVoice> mVoices;
Monitor mVoicesMonitor;
PicoVoice* mCurrentVoice;
Atomic<nsISpeechTask*> mCurrentTask;
pico_System mPicoSystem;
pico_Engine mPicoEngine;
pico_Resource mSgResource;
pico_Resource mTaResource;
mozilla::UniquePtr<uint8_t[]> mPicoMemArea;
static StaticRefPtr<nsPicoService> sSingleton;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,56 @@
/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim: set ts=8 sts=2 et sw=2 tw=80: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "mozilla/ModuleUtils.h"
#include "nsIClassInfoImpl.h"
#include "SpeechDispatcherService.h"
using namespace mozilla::dom;
#define SPEECHDISPATCHERSERVICE_CID \
{0x8817b1cf, 0x5ada, 0x43bf, {0xbd, 0x73, 0x60, 0x76, 0x57, 0x70, 0x3d, 0x0d}}
#define SPEECHDISPATCHERSERVICE_CONTRACTID "@mozilla.org/synthspeechdispatcher;1"
// Defines SpeechDispatcherServiceConstructor
NS_GENERIC_FACTORY_SINGLETON_CONSTRUCTOR(SpeechDispatcherService,
SpeechDispatcherService::GetInstanceForService)
// Defines kSPEECHDISPATCHERSERVICE_CID
NS_DEFINE_NAMED_CID(SPEECHDISPATCHERSERVICE_CID);
static const mozilla::Module::CIDEntry kCIDs[] = {
{ &kSPEECHDISPATCHERSERVICE_CID, true, nullptr, SpeechDispatcherServiceConstructor },
{ nullptr }
};
static const mozilla::Module::ContractIDEntry kContracts[] = {
{ SPEECHDISPATCHERSERVICE_CONTRACTID, &kSPEECHDISPATCHERSERVICE_CID },
{ nullptr }
};
static const mozilla::Module::CategoryEntry kCategories[] = {
{ "speech-synth-started", "SpeechDispatcher Speech Synth", SPEECHDISPATCHERSERVICE_CONTRACTID },
{ nullptr }
};
static void
UnloadSpeechDispatcherModule()
{
SpeechDispatcherService::Shutdown();
}
static const mozilla::Module kModule = {
mozilla::Module::kVersion,
kCIDs,
kContracts,
kCategories,
nullptr,
nullptr,
UnloadSpeechDispatcherModule
};
NSMODULE_DEFN(synthspeechdispatcher) = &kModule;

View file

@ -0,0 +1,593 @@
/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim: set ts=8 sts=2 et sw=2 tw=80: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "SpeechDispatcherService.h"
#include "mozilla/dom/nsSpeechTask.h"
#include "mozilla/dom/nsSynthVoiceRegistry.h"
#include "mozilla/Preferences.h"
#include "nsEscape.h"
#include "nsISupports.h"
#include "nsPrintfCString.h"
#include "nsReadableUtils.h"
#include "nsServiceManagerUtils.h"
#include "nsThreadUtils.h"
#include "prlink.h"
#include <math.h>
#include <stdlib.h>
#define URI_PREFIX "urn:moz-tts:speechd:"
#define MAX_RATE static_cast<float>(2.5)
#define MIN_RATE static_cast<float>(0.5)
// Some structures for libspeechd
typedef enum {
SPD_EVENT_BEGIN,
SPD_EVENT_END,
SPD_EVENT_INDEX_MARK,
SPD_EVENT_CANCEL,
SPD_EVENT_PAUSE,
SPD_EVENT_RESUME
} SPDNotificationType;
typedef enum {
SPD_BEGIN = 1,
SPD_END = 2,
SPD_INDEX_MARKS = 4,
SPD_CANCEL = 8,
SPD_PAUSE = 16,
SPD_RESUME = 32,
SPD_ALL = 0x3f
} SPDNotification;
typedef enum {
SPD_MODE_SINGLE = 0,
SPD_MODE_THREADED = 1
} SPDConnectionMode;
typedef void (*SPDCallback) (size_t msg_id, size_t client_id,
SPDNotificationType state);
typedef void (*SPDCallbackIM) (size_t msg_id, size_t client_id,
SPDNotificationType state, char* index_mark);
struct SPDConnection
{
SPDCallback callback_begin;
SPDCallback callback_end;
SPDCallback callback_cancel;
SPDCallback callback_pause;
SPDCallback callback_resume;
SPDCallbackIM callback_im;
/* partial, more private fields in structure */
};
struct SPDVoice
{
char* name;
char* language;
char* variant;
};
typedef enum {
SPD_IMPORTANT = 1,
SPD_MESSAGE = 2,
SPD_TEXT = 3,
SPD_NOTIFICATION = 4,
SPD_PROGRESS = 5
} SPDPriority;
#define SPEECHD_FUNCTIONS \
FUNC(spd_open, SPDConnection*, (const char*, const char*, const char*, SPDConnectionMode)) \
FUNC(spd_close, void, (SPDConnection*)) \
FUNC(spd_list_synthesis_voices, SPDVoice**, (SPDConnection*)) \
FUNC(spd_say, int, (SPDConnection*, SPDPriority, const char*)) \
FUNC(spd_cancel, int, (SPDConnection*)) \
FUNC(spd_set_volume, int, (SPDConnection*, int)) \
FUNC(spd_set_voice_rate, int, (SPDConnection*, int)) \
FUNC(spd_set_voice_pitch, int, (SPDConnection*, int)) \
FUNC(spd_set_synthesis_voice, int, (SPDConnection*, const char*)) \
FUNC(spd_set_notification_on, int, (SPDConnection*, SPDNotification))
#define FUNC(name, type, params) \
typedef type (*_##name##_fn) params; \
static _##name##_fn _##name;
SPEECHD_FUNCTIONS
#undef FUNC
#define spd_open _spd_open
#define spd_close _spd_close
#define spd_list_synthesis_voices _spd_list_synthesis_voices
#define spd_say _spd_say
#define spd_cancel _spd_cancel
#define spd_set_volume _spd_set_volume
#define spd_set_voice_rate _spd_set_voice_rate
#define spd_set_voice_pitch _spd_set_voice_pitch
#define spd_set_synthesis_voice _spd_set_synthesis_voice
#define spd_set_notification_on _spd_set_notification_on
static PRLibrary* speechdLib = nullptr;
typedef void (*nsSpeechDispatcherFunc)();
struct nsSpeechDispatcherDynamicFunction
{
const char* functionName;
nsSpeechDispatcherFunc* function;
};
namespace mozilla {
namespace dom {
StaticRefPtr<SpeechDispatcherService> SpeechDispatcherService::sSingleton;
class SpeechDispatcherVoice
{
public:
SpeechDispatcherVoice(const nsAString& aName, const nsAString& aLanguage)
: mName(aName), mLanguage(aLanguage) {}
NS_INLINE_DECL_THREADSAFE_REFCOUNTING(SpeechDispatcherVoice)
// Voice name
nsString mName;
// Voice language, in BCP-47 syntax
nsString mLanguage;
private:
~SpeechDispatcherVoice() {}
};
class SpeechDispatcherCallback final : public nsISpeechTaskCallback
{
public:
SpeechDispatcherCallback(nsISpeechTask* aTask, SpeechDispatcherService* aService)
: mTask(aTask)
, mService(aService) {}
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
NS_DECL_CYCLE_COLLECTION_CLASS_AMBIGUOUS(SpeechDispatcherCallback, nsISpeechTaskCallback)
NS_DECL_NSISPEECHTASKCALLBACK
bool OnSpeechEvent(SPDNotificationType state);
private:
~SpeechDispatcherCallback() { }
// This pointer is used to dispatch events
nsCOMPtr<nsISpeechTask> mTask;
// By holding a strong reference to the service we guarantee that it won't be
// destroyed before this runnable.
RefPtr<SpeechDispatcherService> mService;
TimeStamp mStartTime;
};
NS_IMPL_CYCLE_COLLECTION(SpeechDispatcherCallback, mTask);
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(SpeechDispatcherCallback)
NS_INTERFACE_MAP_ENTRY(nsISpeechTaskCallback)
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsISpeechTaskCallback)
NS_INTERFACE_MAP_END
NS_IMPL_CYCLE_COLLECTING_ADDREF(SpeechDispatcherCallback)
NS_IMPL_CYCLE_COLLECTING_RELEASE(SpeechDispatcherCallback)
NS_IMETHODIMP
SpeechDispatcherCallback::OnPause()
{
// XXX: Speech dispatcher does not pause immediately, but waits for the speech
// to reach an index mark so that it could resume from that offset.
// There is no support for word or sentence boundaries, so index marks would
// only occur in explicit SSML marks, and we don't support that yet.
// What in actuality happens, is that if you call spd_pause(), it will speak
// the utterance in its entirety, dispatch an end event, and then put speechd
// in a 'paused' state. Since it is after the utterance ended, we don't get
// that state change, and our speech api is in an unrecoverable state.
// So, since it is useless anyway, I am not implementing pause.
return NS_OK;
}
NS_IMETHODIMP
SpeechDispatcherCallback::OnResume()
{
// XXX: Unsupported, see OnPause().
return NS_OK;
}
NS_IMETHODIMP
SpeechDispatcherCallback::OnCancel()
{
if (spd_cancel(mService->mSpeechdClient) < 0) {
return NS_ERROR_FAILURE;
}
return NS_OK;
}
NS_IMETHODIMP
SpeechDispatcherCallback::OnVolumeChanged(float aVolume)
{
// XXX: This currently does not change the volume mid-utterance, but it
// doesn't do anything bad either. So we could put this here with the hopes
// that speechd supports this in the future.
if (spd_set_volume(mService->mSpeechdClient, static_cast<int>(aVolume * 100)) < 0) {
return NS_ERROR_FAILURE;
}
return NS_OK;
}
bool
SpeechDispatcherCallback::OnSpeechEvent(SPDNotificationType state)
{
bool remove = false;
switch (state) {
case SPD_EVENT_BEGIN:
mStartTime = TimeStamp::Now();
mTask->DispatchStart();
break;
case SPD_EVENT_PAUSE:
mTask->DispatchPause((TimeStamp::Now() - mStartTime).ToSeconds(), 0);
break;
case SPD_EVENT_RESUME:
mTask->DispatchResume((TimeStamp::Now() - mStartTime).ToSeconds(), 0);
break;
case SPD_EVENT_CANCEL:
case SPD_EVENT_END:
mTask->DispatchEnd((TimeStamp::Now() - mStartTime).ToSeconds(), 0);
remove = true;
break;
case SPD_EVENT_INDEX_MARK:
// Not yet supported
break;
default:
break;
}
return remove;
}
static void
speechd_cb(size_t msg_id, size_t client_id, SPDNotificationType state)
{
SpeechDispatcherService* service = SpeechDispatcherService::GetInstance(false);
if (service) {
NS_DispatchToMainThread(
NewRunnableMethod<uint32_t, SPDNotificationType>(
service, &SpeechDispatcherService::EventNotify,
static_cast<uint32_t>(msg_id), state));
}
}
NS_INTERFACE_MAP_BEGIN(SpeechDispatcherService)
NS_INTERFACE_MAP_ENTRY(nsISpeechService)
NS_INTERFACE_MAP_ENTRY(nsIObserver)
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsIObserver)
NS_INTERFACE_MAP_END
NS_IMPL_ADDREF(SpeechDispatcherService)
NS_IMPL_RELEASE(SpeechDispatcherService)
SpeechDispatcherService::SpeechDispatcherService()
: mInitialized(false)
, mSpeechdClient(nullptr)
{
}
void
SpeechDispatcherService::Init()
{
if (!Preferences::GetBool("media.webspeech.synth.enabled") ||
Preferences::GetBool("media.webspeech.synth.test")) {
return;
}
// While speech dispatcher has a "threaded" mode, only spd_say() is async.
// Since synchronous socket i/o could impact startup time, we do
// initialization in a separate thread.
DebugOnly<nsresult> rv = NS_NewNamedThread("speechd init",
getter_AddRefs(mInitThread));
MOZ_ASSERT(NS_SUCCEEDED(rv));
rv = mInitThread->Dispatch(
NewRunnableMethod(this, &SpeechDispatcherService::Setup), NS_DISPATCH_NORMAL);
MOZ_ASSERT(NS_SUCCEEDED(rv));
}
SpeechDispatcherService::~SpeechDispatcherService()
{
if (mInitThread) {
mInitThread->Shutdown();
}
if (mSpeechdClient) {
spd_close(mSpeechdClient);
}
}
void
SpeechDispatcherService::Setup()
{
#define FUNC(name, type, params) { #name, (nsSpeechDispatcherFunc *)&_##name },
static const nsSpeechDispatcherDynamicFunction kSpeechDispatcherSymbols[] = {
SPEECHD_FUNCTIONS
};
#undef FUNC
MOZ_ASSERT(!mInitialized);
speechdLib = PR_LoadLibrary("libspeechd.so.2");
if (!speechdLib) {
NS_WARNING("Failed to load speechd library");
return;
}
if (!PR_FindFunctionSymbol(speechdLib, "spd_get_volume")) {
// There is no version getter function, so we rely on a symbol that was
// introduced in release 0.8.2 in order to check for ABI compatibility.
NS_WARNING("Unsupported version of speechd detected");
return;
}
for (uint32_t i = 0; i < ArrayLength(kSpeechDispatcherSymbols); i++) {
*kSpeechDispatcherSymbols[i].function =
PR_FindFunctionSymbol(speechdLib, kSpeechDispatcherSymbols[i].functionName);
if (!*kSpeechDispatcherSymbols[i].function) {
NS_WARNING(nsPrintfCString("Failed to find speechd symbol for'%s'",
kSpeechDispatcherSymbols[i].functionName).get());
return;
}
}
mSpeechdClient = spd_open("firefox", "web speech api", "who", SPD_MODE_THREADED);
if (!mSpeechdClient) {
NS_WARNING("Failed to call spd_open");
return;
}
// Get all the voices from sapi and register in the SynthVoiceRegistry
SPDVoice** list = spd_list_synthesis_voices(mSpeechdClient);
mSpeechdClient->callback_begin = speechd_cb;
mSpeechdClient->callback_end = speechd_cb;
mSpeechdClient->callback_cancel = speechd_cb;
mSpeechdClient->callback_pause = speechd_cb;
mSpeechdClient->callback_resume = speechd_cb;
spd_set_notification_on(mSpeechdClient, SPD_BEGIN);
spd_set_notification_on(mSpeechdClient, SPD_END);
spd_set_notification_on(mSpeechdClient, SPD_CANCEL);
if (list != NULL) {
for (int i = 0; list[i]; i++) {
nsAutoString uri;
uri.AssignLiteral(URI_PREFIX);
nsAutoCString name;
NS_EscapeURL(list[i]->name, -1, esc_OnlyNonASCII | esc_AlwaysCopy, name);
uri.Append(NS_ConvertUTF8toUTF16(name));;
uri.AppendLiteral("?");
nsAutoCString lang(list[i]->language);
if (strcmp(list[i]->variant, "none") != 0) {
// In speech dispatcher, the variant will usually be the locale subtag
// with another, non-standard suptag after it. We keep the first one
// and convert it to uppercase.
const char* v = list[i]->variant;
const char* hyphen = strchr(v, '-');
nsDependentCSubstring variant(v, hyphen ? hyphen - v : strlen(v));
ToUpperCase(variant);
// eSpeak uses UK which is not a valid region subtag in BCP47.
if (variant.Equals("UK")) {
variant.AssignLiteral("GB");
}
lang.AppendLiteral("-");
lang.Append(variant);
}
uri.Append(NS_ConvertUTF8toUTF16(lang));
mVoices.Put(uri, new SpeechDispatcherVoice(
NS_ConvertUTF8toUTF16(list[i]->name),
NS_ConvertUTF8toUTF16(lang)));
}
}
NS_DispatchToMainThread(NewRunnableMethod(this, &SpeechDispatcherService::RegisterVoices));
//mInitialized = true;
}
// private methods
void
SpeechDispatcherService::RegisterVoices()
{
RefPtr<nsSynthVoiceRegistry> registry = nsSynthVoiceRegistry::GetInstance();
for (auto iter = mVoices.Iter(); !iter.Done(); iter.Next()) {
RefPtr<SpeechDispatcherVoice>& voice = iter.Data();
// This service can only speak one utterance at a time, so we set
// aQueuesUtterances to true in order to track global state and schedule
// access to this service.
DebugOnly<nsresult> rv =
registry->AddVoice(this, iter.Key(), voice->mName, voice->mLanguage,
voice->mName.EqualsLiteral("default"), true);
NS_WARNING_ASSERTION(NS_SUCCEEDED(rv), "Failed to add voice");
}
mInitThread->Shutdown();
mInitThread = nullptr;
mInitialized = true;
registry->NotifyVoicesChanged();
}
// nsIObserver
NS_IMETHODIMP
SpeechDispatcherService::Observe(nsISupports* aSubject, const char* aTopic,
const char16_t* aData)
{
return NS_OK;
}
// nsISpeechService
// TODO: Support SSML
NS_IMETHODIMP
SpeechDispatcherService::Speak(const nsAString& aText, const nsAString& aUri,
float aVolume, float aRate, float aPitch,
nsISpeechTask* aTask)
{
if (NS_WARN_IF(!mInitialized)) {
return NS_ERROR_NOT_AVAILABLE;
}
RefPtr<SpeechDispatcherCallback> callback =
new SpeechDispatcherCallback(aTask, this);
bool found = false;
SpeechDispatcherVoice* voice = mVoices.GetWeak(aUri, &found);
if(NS_WARN_IF(!(found))) {
return NS_ERROR_NOT_AVAILABLE;
}
spd_set_synthesis_voice(mSpeechdClient,
NS_ConvertUTF16toUTF8(voice->mName).get());
// We provide a volume of 0.0 to 1.0, speech-dispatcher expects 0 - 100.
spd_set_volume(mSpeechdClient, static_cast<int>(aVolume * 100));
// aRate is a value of 0.1 (0.1x) to 10 (10x) with 1 (1x) being normal rate.
// speechd expects -100 to 100 with 0 being normal rate.
float rate = 0;
if (aRate > 1) {
// Each step to 100 is logarithmically distributed up to 2.5x.
rate = log10(std::min(aRate, MAX_RATE)) / log10(MAX_RATE) * 100;
} else if (aRate < 1) {
// Each step to -100 is logarithmically distributed down to 0.5x.
rate = log10(std::max(aRate, MIN_RATE)) / log10(MIN_RATE) * -100;
}
spd_set_voice_rate(mSpeechdClient, static_cast<int>(rate));
// We provide a pitch of 0 to 2 with 1 being the default.
// speech-dispatcher expects -100 to 100 with 0 being default.
spd_set_voice_pitch(mSpeechdClient, static_cast<int>((aPitch - 1) * 100));
// The last three parameters don't matter for an indirect service
nsresult rv = aTask->Setup(callback, 0, 0, 0);
if (NS_FAILED(rv)) {
return rv;
}
if (aText.Length()) {
int msg_id = spd_say(
mSpeechdClient, SPD_MESSAGE, NS_ConvertUTF16toUTF8(aText).get());
if (msg_id < 0) {
return NS_ERROR_FAILURE;
}
mCallbacks.Put(msg_id, callback);
} else {
// Speech dispatcher does not work well with empty strings.
// In that case, don't send empty string to speechd,
// and just emulate a speechd start and end event.
NS_DispatchToMainThread(NewRunnableMethod<SPDNotificationType>(
callback, &SpeechDispatcherCallback::OnSpeechEvent, SPD_EVENT_BEGIN));
NS_DispatchToMainThread(NewRunnableMethod<SPDNotificationType>(
callback, &SpeechDispatcherCallback::OnSpeechEvent, SPD_EVENT_END));
}
return NS_OK;
}
NS_IMETHODIMP
SpeechDispatcherService::GetServiceType(SpeechServiceType* aServiceType)
{
*aServiceType = nsISpeechService::SERVICETYPE_INDIRECT_AUDIO;
return NS_OK;
}
SpeechDispatcherService*
SpeechDispatcherService::GetInstance(bool create)
{
if (XRE_GetProcessType() != GeckoProcessType_Default) {
MOZ_ASSERT(false,
"SpeechDispatcherService can only be started on main gecko process");
return nullptr;
}
if (!sSingleton && create) {
sSingleton = new SpeechDispatcherService();
sSingleton->Init();
}
return sSingleton;
}
already_AddRefed<SpeechDispatcherService>
SpeechDispatcherService::GetInstanceForService()
{
MOZ_ASSERT(NS_IsMainThread());
RefPtr<SpeechDispatcherService> sapiService = GetInstance();
return sapiService.forget();
}
void
SpeechDispatcherService::EventNotify(uint32_t aMsgId, uint32_t aState)
{
SpeechDispatcherCallback* callback = mCallbacks.GetWeak(aMsgId);
if (callback) {
if (callback->OnSpeechEvent((SPDNotificationType)aState)) {
mCallbacks.Remove(aMsgId);
}
}
}
void
SpeechDispatcherService::Shutdown()
{
if (!sSingleton) {
return;
}
sSingleton = nullptr;
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,67 @@
/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim: set ts=8 sts=2 et sw=2 tw=80: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef mozilla_dom_SpeechDispatcherService_h
#define mozilla_dom_SpeechDispatcherService_h
#include "mozilla/StaticPtr.h"
#include "nsIObserver.h"
#include "nsISpeechService.h"
#include "nsIThread.h"
#include "nsRefPtrHashtable.h"
#include "nsTArray.h"
struct SPDConnection;
namespace mozilla {
namespace dom {
class SpeechDispatcherCallback;
class SpeechDispatcherVoice;
class SpeechDispatcherService final : public nsIObserver,
public nsISpeechService
{
friend class SpeechDispatcherCallback;
public:
NS_DECL_THREADSAFE_ISUPPORTS
NS_DECL_NSIOBSERVER
NS_DECL_NSISPEECHSERVICE
SpeechDispatcherService();
void Init();
void Setup();
void EventNotify(uint32_t aMsgId, uint32_t aState);
static SpeechDispatcherService* GetInstance(bool create = true);
static already_AddRefed<SpeechDispatcherService> GetInstanceForService();
static void Shutdown();
static StaticRefPtr<SpeechDispatcherService> sSingleton;
private:
virtual ~SpeechDispatcherService();
void RegisterVoices();
bool mInitialized;
SPDConnection* mSpeechdClient;
nsRefPtrHashtable<nsUint32HashKey, SpeechDispatcherCallback> mCallbacks;
nsCOMPtr<nsIThread> mInitThread;
nsRefPtrHashtable<nsStringHashKey, SpeechDispatcherVoice> mVoices;
};
} // namespace dom
} // namespace mozilla
#endif

View file

@ -0,0 +1,13 @@
# -*- Mode: python; indent-tabs-mode: nil; tab-width: 40 -*-
# vim: set filetype=python:
# This Source Code Form is subject to the terms of the Mozilla Public
# License, v. 2.0. If a copy of the MPL was not distributed with this
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
UNIFIED_SOURCES += [
'SpeechDispatcherModule.cpp',
'SpeechDispatcherService.cpp'
]
include('/ipc/chromium/chromium-config.mozbuild')
FINAL_LIBRARY = 'xul'

View file

@ -0,0 +1,55 @@
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "mozilla/ModuleUtils.h"
#include "nsIClassInfoImpl.h"
#include "nsFakeSynthServices.h"
using namespace mozilla::dom;
#define FAKESYNTHSERVICE_CID \
{0xe7d52d9e, 0xc148, 0x47d8, {0xab, 0x2a, 0x95, 0xd7, 0xf4, 0x0e, 0xa5, 0x3d}}
#define FAKESYNTHSERVICE_CONTRACTID "@mozilla.org/fakesynth;1"
// Defines nsFakeSynthServicesConstructor
NS_GENERIC_FACTORY_SINGLETON_CONSTRUCTOR(nsFakeSynthServices,
nsFakeSynthServices::GetInstanceForService)
// Defines kFAKESYNTHSERVICE_CID
NS_DEFINE_NAMED_CID(FAKESYNTHSERVICE_CID);
static const mozilla::Module::CIDEntry kCIDs[] = {
{ &kFAKESYNTHSERVICE_CID, true, nullptr, nsFakeSynthServicesConstructor },
{ nullptr }
};
static const mozilla::Module::ContractIDEntry kContracts[] = {
{ FAKESYNTHSERVICE_CONTRACTID, &kFAKESYNTHSERVICE_CID },
{ nullptr }
};
static const mozilla::Module::CategoryEntry kCategories[] = {
{ "speech-synth-started", "Fake Speech Synth", FAKESYNTHSERVICE_CONTRACTID },
{ nullptr }
};
static void
UnloadFakeSynthmodule()
{
nsFakeSynthServices::Shutdown();
}
static const mozilla::Module kModule = {
mozilla::Module::kVersion,
kCIDs,
kContracts,
kCategories,
nullptr,
nullptr,
UnloadFakeSynthmodule
};
NSMODULE_DEFN(fakesynth) = &kModule;

View file

@ -0,0 +1,91 @@
function synthTestQueue(aTestArgs, aEndFunc) {
var utterances = [];
for (var i in aTestArgs) {
var uargs = aTestArgs[i][0];
var win = uargs.win || window;
var u = new win.SpeechSynthesisUtterance(uargs.text);
if (uargs.args) {
for (var attr in uargs.args)
u[attr] = uargs.args[attr];
}
function onend_handler(e) {
is(e.target, utterances.shift(), "Target matches utterances");
ok(!speechSynthesis.speaking, "speechSynthesis is not speaking.");
if (utterances.length) {
ok(speechSynthesis.pending, "other utterances queued");
} else {
ok(!speechSynthesis.pending, "queue is empty, nothing pending.");
if (aEndFunc)
aEndFunc();
}
}
u.addEventListener('start',
(function (expectedUri) {
return function (e) {
if (expectedUri) {
var chosenVoice = SpecialPowers.wrap(e).target.chosenVoiceURI;
is(chosenVoice, expectedUri, "Incorrect URI is used");
}
};
})(aTestArgs[i][1] ? aTestArgs[i][1].uri : null));
u.addEventListener('end', onend_handler);
u.addEventListener('error', onend_handler);
u.addEventListener('error',
(function (expectedError) {
return function onerror_handler(e) {
ok(expectedError, "Error in speech utterance '" + e.target.text + "'");
};
})(aTestArgs[i][1] ? aTestArgs[i][1].err : false));
utterances.push(u);
win.speechSynthesis.speak(u);
}
ok(!speechSynthesis.speaking, "speechSynthesis is not speaking yet.");
ok(speechSynthesis.pending, "speechSynthesis has an utterance queued.");
}
function loadFrame(frameId) {
return new Promise(function(resolve, reject) {
var frame = document.getElementById(frameId);
frame.addEventListener('load', function (e) {
frame.contentWindow.document.title = frameId;
resolve(frame);
});
frame.src = 'data:text/html,' + encodeURI('<html><head></head><body></body></html>');
});
}
function waitForVoices(win) {
return new Promise(resolve => {
function resolver() {
if (win.speechSynthesis.getVoices().length) {
win.speechSynthesis.removeEventListener('voiceschanged', resolver);
resolve();
}
}
win.speechSynthesis.addEventListener('voiceschanged', resolver);
resolver();
});
}
function loadSpeechTest(fileName, prefs, frameId="testFrame") {
loadFrame(frameId).then(frame => {
waitForVoices(frame.contentWindow).then(
() => document.getElementById("testFrame").src = fileName);
});
}
function testSynthState(win, expectedState) {
for (var attr in expectedState) {
is(win.speechSynthesis[attr], expectedState[attr],
win.document.title + ": '" + attr + '" does not match');
}
}

View file

@ -0,0 +1,28 @@
<!DOCTYPE HTML>
<html>
<head>
<meta charset="utf-8">
<script type="application/javascript">
var frameUnloaded = function() {
var u = new SpeechSynthesisUtterance('hi');
u.addEventListener('end', function () {
parent.ok(true, 'Successfully spoke utterance from new frame.');
parent.onDone();
});
speechSynthesis.speak(u);
}
addEventListener('pageshow', function onshow(evt) {
var u = new SpeechSynthesisUtterance('hello');
u.lang = 'it-IT-noend';
u.addEventListener('start', function() {
location =
'data:text/html,<html><body onpageshow="' +
frameUnloaded.toSource() + '()"></body></html>';
});
speechSynthesis.speak(u);
});
</script>
</head>
<body>
</body>
</html>

View file

@ -0,0 +1,69 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=1188099
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 1188099: Global queue should correctly schedule utterances</title>
<script type="application/javascript">
window.SimpleTest = parent.SimpleTest;
window.info = parent.info;
window.is = parent.is;
window.isnot = parent.isnot;
window.ok = parent.ok;
window.todo = parent.todo;
</script>
<script type="application/javascript" src="common.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1188099">Mozilla Bug 1188099</a>
<iframe id="frame1"></iframe>
<iframe id="frame2"></iframe>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="application/javascript">
Promise.all([loadFrame('frame1'), loadFrame('frame2')]).then(function ([frame1, frame2]) {
var win1 = frame1.contentWindow;
var win2 = frame2.contentWindow;
var utterance1 = new win1.SpeechSynthesisUtterance("hello, losers");
var utterance2 = new win1.SpeechSynthesisUtterance("hello, losers three");
var utterance3 = new win2.SpeechSynthesisUtterance("hello, losers too");
var eventOrder = ['start1', 'end1', 'start3', 'end3', 'start2', 'end2'];
utterance1.addEventListener('start', function(e) {
is(eventOrder.shift(), 'start1', 'start1');
testSynthState(win1, { speaking: true, pending: true });
testSynthState(win2, { speaking: true, pending: true });
});
utterance1.addEventListener('end', function(e) {
is(eventOrder.shift(), 'end1', 'end1');
});
utterance3.addEventListener('start', function(e) {
is(eventOrder.shift(), 'start3', 'start3');
testSynthState(win1, { speaking: true, pending: true });
testSynthState(win2, { speaking: true, pending: false });
});
utterance3.addEventListener('end', function(e) {
is(eventOrder.shift(), 'end3', 'end3');
});
utterance2.addEventListener('start', function(e) {
is(eventOrder.shift(), 'start2', 'start2');
testSynthState(win1, { speaking: true, pending: false });
testSynthState(win2, { speaking: true, pending: false });
});
utterance2.addEventListener('end', function(e) {
is(eventOrder.shift(), 'end2', 'end2');
testSynthState(win1, { speaking: false, pending: false });
testSynthState(win2, { speaking: false, pending: false });
SimpleTest.finish();
});
win1.speechSynthesis.speak(utterance1);
win1.speechSynthesis.speak(utterance2);
win2.speechSynthesis.speak(utterance3);
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,88 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=1188099
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 1188099: Calling cancel() should work correctly with global queue</title>
<script type="application/javascript">
window.SimpleTest = parent.SimpleTest;
window.info = parent.info;
window.is = parent.is;
window.isnot = parent.isnot;
window.ok = parent.ok;
window.todo = parent.todo;
</script>
<script type="application/javascript" src="common.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1188099">Mozilla Bug 1188099</a>
<iframe id="frame1"></iframe>
<iframe id="frame2"></iframe>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="application/javascript">
Promise.all([loadFrame('frame1'), loadFrame('frame2')]).then(function ([frame1, frame2]) {
var win1 = frame1.contentWindow;
var win2 = frame2.contentWindow;
var utterance1 = new win1.SpeechSynthesisUtterance(
"u1: Donec ac nunc feugiat, posuere");
utterance1.lang = 'it-IT-noend';
var utterance2 = new win1.SpeechSynthesisUtterance("u2: hello, losers too");
utterance2.lang = 'it-IT-noend';
var utterance3 = new win1.SpeechSynthesisUtterance("u3: hello, losers three");
var utterance4 = new win2.SpeechSynthesisUtterance("u4: hello, losers same!");
utterance4.lang = 'it-IT-noend';
var utterance5 = new win2.SpeechSynthesisUtterance("u5: hello, losers too");
utterance5.lang = 'it-IT-noend';
var eventOrder = ['start1', 'end1', 'start2', 'end2'];
utterance1.addEventListener('start', function(e) {
is(eventOrder.shift(), 'start1', 'start1');
testSynthState(win1, { speaking: true, pending: true });
testSynthState(win2, { speaking: true, pending: true });
win2.speechSynthesis.cancel();
SpecialPowers.wrap(win1.speechSynthesis).forceEnd();
});
utterance1.addEventListener('end', function(e) {
is(eventOrder.shift(), 'end1', 'end1');
testSynthState(win1, { pending: true });
testSynthState(win2, { pending: false });
});
utterance2.addEventListener('start', function(e) {
is(eventOrder.shift(), 'start2', 'start2');
testSynthState(win1, { speaking: true, pending: true });
testSynthState(win2, { speaking: true, pending: false });
win1.speechSynthesis.cancel();
});
utterance2.addEventListener('end', function(e) {
is(eventOrder.shift(), 'end2', 'end2');
testSynthState(win1, { speaking: false, pending: false });
testSynthState(win2, { speaking: false, pending: false });
SimpleTest.finish();
});
function wrongUtterance(e) {
ok(false, 'This shall not be uttered: "' + e.target.text + '"');
}
utterance3.addEventListener('start', wrongUtterance);
utterance4.addEventListener('start', wrongUtterance);
utterance5.addEventListener('start', wrongUtterance);
win1.speechSynthesis.speak(utterance1);
win1.speechSynthesis.speak(utterance2);
win1.speechSynthesis.speak(utterance3);
win2.speechSynthesis.speak(utterance4);
win2.speechSynthesis.speak(utterance5);
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,131 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=1188099
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 1188099: Calling pause() should work correctly with global queue</title>
<script type="application/javascript">
window.SimpleTest = parent.SimpleTest;
window.info = parent.info;
window.is = parent.is;
window.isnot = parent.isnot;
window.ok = parent.ok;
window.todo = parent.todo;
</script>
<script type="application/javascript" src="common.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1188099">Mozilla Bug 1188099</a>
<iframe id="frame1"></iframe>
<iframe id="frame2"></iframe>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="application/javascript">
Promise.all([loadFrame('frame1'), loadFrame('frame2')]).then(function ([frame1, frame2]) {
var win1 = frame1.contentWindow;
var win2 = frame2.contentWindow;
var utterance1 = new win1.SpeechSynthesisUtterance("Speak utterance 1.");
utterance1.lang = 'it-IT-noend';
var utterance2 = new win2.SpeechSynthesisUtterance("Speak utterance 2.");
var utterance3 = new win1.SpeechSynthesisUtterance("Speak utterance 3.");
var utterance4 = new win2.SpeechSynthesisUtterance("Speak utterance 4.");
var eventOrder = ['start1', 'pause1', 'resume1', 'end1', 'start2', 'end2',
'start4', 'end4', 'start3', 'end3'];
utterance1.addEventListener('start', function(e) {
is(eventOrder.shift(), 'start1', 'start1');
win1.speechSynthesis.pause();
});
utterance1.addEventListener('pause', function(e) {
var expectedEvent = eventOrder.shift()
is(expectedEvent, 'pause1', 'pause1');
testSynthState(win1, { speaking: true, pending: false, paused: true});
testSynthState(win2, { speaking: true, pending: true, paused: false});
if (expectedEvent == 'pause1') {
win1.speechSynthesis.resume();
}
});
utterance1.addEventListener('resume', function(e) {
is(eventOrder.shift(), 'resume1', 'resume1');
testSynthState(win1, { speaking: true, pending: false, paused: false});
testSynthState(win2, { speaking: true, pending: true, paused: false});
win2.speechSynthesis.pause();
testSynthState(win1, { speaking: true, pending: false, paused: false});
// 1188099: currently, paused state is not gaurenteed to be immediate.
testSynthState(win2, { speaking: true, pending: true });
// We now make the utterance end.
SpecialPowers.wrap(win1.speechSynthesis).forceEnd();
});
utterance1.addEventListener('end', function(e) {
is(eventOrder.shift(), 'end1', 'end1');
testSynthState(win1, { speaking: false, pending: false, paused: false});
testSynthState(win2, { speaking: false, pending: true, paused: true});
win2.speechSynthesis.resume();
});
utterance2.addEventListener('start', function(e) {
is(eventOrder.shift(), 'start2', 'start2');
testSynthState(win1, { speaking: true, pending: false, paused: false});
testSynthState(win2, { speaking: true, pending: false, paused: false});
});
utterance2.addEventListener('end', function(e) {
is(eventOrder.shift(), 'end2', 'end2');
testSynthState(win1, { speaking: false, pending: false, paused: false});
testSynthState(win2, { speaking: false, pending: false, paused: false});
win1.speechSynthesis.pause();
testSynthState(win1, { speaking: false, pending: false, paused: true});
testSynthState(win2, { speaking: false, pending: false, paused: false});
win1.speechSynthesis.speak(utterance3);
win2.speechSynthesis.speak(utterance4);
testSynthState(win1, { speaking: false, pending: true, paused: true});
testSynthState(win2, { speaking: false, pending: true, paused: false});
});
utterance4.addEventListener('start', function(e) {
is(eventOrder.shift(), 'start4', 'start4');
testSynthState(win1, { speaking: true, pending: true, paused: true});
testSynthState(win2, { speaking: true, pending: false, paused: false});
win1.speechSynthesis.resume();
});
utterance4.addEventListener('end', function(e) {
is(eventOrder.shift(), 'end4', 'end4');
testSynthState(win1, { speaking: false, pending: true, paused: false});
testSynthState(win2, { speaking: false, pending: false, paused: false});
});
utterance3.addEventListener('start', function(e) {
is(eventOrder.shift(), 'start3', 'start3');
testSynthState(win1, { speaking: true, pending: false, paused: false});
testSynthState(win2, { speaking: true, pending: false, paused: false});
});
utterance3.addEventListener('end', function(e) {
is(eventOrder.shift(), 'end3', 'end3');
testSynthState(win1, { speaking: false, pending: false, paused: false});
testSynthState(win2, { speaking: false, pending: false, paused: false});
SimpleTest.finish();
});
win1.speechSynthesis.speak(utterance1);
win2.speechSynthesis.speak(utterance2);
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,102 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=1155034
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 1155034: Check that indirect audio services dispatch their own events</title>
<script type="application/javascript">
window.SimpleTest = parent.SimpleTest;
window.info = parent.info;
window.is = parent.is;
window.isnot = parent.isnot;
window.ok = parent.ok;
</script>
<script type="application/javascript" src="common.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1155034">Mozilla Bug 1155034</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="application/javascript">
/** Test for Bug 1155034 **/
function testFunc(done_cb) {
function test_with_events() {
info('test_with_events');
var utterance = new SpeechSynthesisUtterance("never end, callback events");
utterance.lang = 'it-IT-noend';
utterance.addEventListener('start', function(e) {
info('start test_with_events');
speechSynthesis.pause();
// Wait to see if we get some bad events we didn't expect.
});
utterance.addEventListener('pause', function(e) {
is(e.charIndex, 1, 'pause event charIndex matches service arguments');
is(e.elapsedTime, 1.5, 'pause event elapsedTime matches service arguments');
speechSynthesis.resume();
});
utterance.addEventListener('resume', function(e) {
is(e.charIndex, 1, 'resume event charIndex matches service arguments');
is(e.elapsedTime, 1.5, 'resume event elapsedTime matches service arguments');
speechSynthesis.cancel();
});
utterance.addEventListener('end', function(e) {
ok(e.charIndex, 1, 'resume event charIndex matches service arguments');
ok(e.elapsedTime, 1.5, 'end event elapsedTime matches service arguments');
test_no_events();
});
info('start speak');
speechSynthesis.speak(utterance);
}
function forbiddenEvent(e) {
ok(false, 'no "' + e.type + '" event was explicitly dispatched from the service')
}
function test_no_events() {
info('test_no_events');
var utterance = new SpeechSynthesisUtterance("never end");
utterance.lang = "it-IT-noevents-noend";
utterance.addEventListener('start', function(e) {
speechSynthesis.pause();
// Wait to see if we get some bad events we didn't expect.
setTimeout(function() {
ok(true, 'didn\'t get any unwanted events');
utterance.removeEventListener('end', forbiddenEvent);
SpecialPowers.wrap(speechSynthesis).forceEnd();
done_cb();
}, 1000);
});
utterance.addEventListener('pause', forbiddenEvent);
utterance.addEventListener('end', forbiddenEvent);
speechSynthesis.speak(utterance);
}
test_with_events();
}
// Run test with no global queue, and then run it with a global queue.
testFunc(function() {
SpecialPowers.pushPrefEnv(
{ set: [['media.webspeech.synth.force_global_queue', true]] }, function() {
testFunc(SimpleTest.finish)
});
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,95 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=525444
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 525444: Web Speech API check all classes are present</title>
<script type="application/javascript">
window.SimpleTest = parent.SimpleTest;
window.is = parent.is;
window.isnot = parent.isnot;
window.ok = parent.ok;
</script>
<script type="application/javascript" src="common.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="application/javascript">
/** Test for Bug 525444 **/
ok(SpeechSynthesis, "SpeechSynthesis exists in global scope");
ok(SpeechSynthesisVoice, "SpeechSynthesisVoice exists in global scope");
ok(SpeechSynthesisErrorEvent, "SpeechSynthesisErrorEvent exists in global scope");
ok(SpeechSynthesisEvent, "SpeechSynthesisEvent exists in global scope");
// SpeechSynthesisUtterance is the only type that has a constructor
// and writable properties
ok(SpeechSynthesisUtterance, "SpeechSynthesisUtterance exists in global scope");
var ssu = new SpeechSynthesisUtterance("hello world");
is(typeof ssu, "object", "SpeechSynthesisUtterance instance is an object");
is(ssu.text, "hello world", "SpeechSynthesisUtterance.text is correct");
is(ssu.volume, 1, "SpeechSynthesisUtterance.volume default is correct");
is(ssu.rate, 1, "SpeechSynthesisUtterance.rate default is correct");
is(ssu.pitch, 1, "SpeechSynthesisUtterance.pitch default is correct");
ssu.lang = "he-IL";
ssu.volume = 0.5;
ssu.rate = 2.0;
ssu.pitch = 1.5;
is(ssu.lang, "he-IL", "SpeechSynthesisUtterance.lang is correct");
is(ssu.volume, 0.5, "SpeechSynthesisUtterance.volume is correct");
is(ssu.rate, 2.0, "SpeechSynthesisUtterance.rate is correct");
is(ssu.pitch, 1.5, "SpeechSynthesisUtterance.pitch is correct");
// Assign a rate that is out of bounds
ssu.rate = 20;
is(ssu.rate, 10, "SpeechSynthesisUtterance.rate enforces max of 10");
ssu.rate = 0;
is(ssu.rate.toPrecision(1), "0.1", "SpeechSynthesisUtterance.rate enforces min of 0.1");
// Assign a volume which is out of bounds
ssu.volume = 2;
is(ssu.volume, 1, "SpeechSynthesisUtterance.volume enforces max of 1");
ssu.volume = -1;
is(ssu.volume, 0, "SpeechSynthesisUtterance.volume enforces min of 0");
// Assign a pitch which is out of bounds
ssu.pitch = 2.1;
is(ssu.pitch, 2, "SpeechSynthesisUtterance.pitch enforces max of 2");
ssu.pitch = -1;
is(ssu.pitch, 0, "SpeechSynthesisUtterance.pitch enforces min of 0");
// Test for singleton instance hanging off of window.
ok(speechSynthesis, "speechSynthesis exists in global scope");
is(typeof speechSynthesis, "object", "speechSynthesis instance is an object");
is(typeof speechSynthesis.speak, "function", "speechSynthesis.speak is a function");
is(typeof speechSynthesis.cancel, "function", "speechSynthesis.cancel is a function");
is(typeof speechSynthesis.pause, "function", "speechSynthesis.pause is a function");
is(typeof speechSynthesis.resume, "function", "speechSynthesis.resume is a function");
is(typeof speechSynthesis.getVoices, "function", "speechSynthesis.getVoices is a function");
is(typeof speechSynthesis.pending, "boolean", "speechSynthesis.pending is a boolean");
is(typeof speechSynthesis.speaking, "boolean", "speechSynthesis.speaking is a boolean");
is(typeof speechSynthesis.paused, "boolean", "speechSynthesis.paused is a boolean");
var voices1 = speechSynthesis.getVoices();
var voices2 = speechSynthesis.getVoices();
ok(voices1.length == voices2.length, "Voice count matches");
for (var i in voices1) {
ok(voices1[i] == voices2[i], "Voice instance matches");
}
SimpleTest.finish();
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,100 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=1150315
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 1150315: Check that successive cancel/speak calls work</title>
<script type="application/javascript">
window.SimpleTest = parent.SimpleTest;
window.info = parent.info;
window.is = parent.is;
window.isnot = parent.isnot;
window.ok = parent.ok;
</script>
<script type="application/javascript" src="common.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1150315">Mozilla Bug 1150315</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="application/javascript">
/** Test for Bug 1150315 **/
function testFunc(done_cb) {
var gotEndEvent = false;
// A long utterance that we will interrupt.
var utterance = new SpeechSynthesisUtterance("Donec ac nunc feugiat, posuere " +
"mauris id, pharetra velit. Donec fermentum orci nunc, sit amet maximus" +
"dui tincidunt ut. Sed ultricies ac nisi a laoreet. Proin interdum," +
"libero maximus hendrerit posuere, lorem risus egestas nisl, a" +
"ultricies massa justo eu nisi. Duis mattis nibh a ligula tincidunt" +
"tincidunt non eu erat. Sed bibendum varius vulputate. Cras leo magna," +
"ornare ac posuere vel, luctus id metus. Mauris nec quam ac augue" +
"consectetur bibendum. Integer a commodo tortor. Duis semper dolor eu" +
"facilisis facilisis. Etiam venenatis turpis est, quis tincidunt velit" +
"suscipit a. Cras semper orci in sapien rhoncus bibendum. Suspendisse" +
"eu ex lobortis, finibus enim in, condimentum quam. Maecenas eget dui" +
"ipsum. Aliquam tortor leo, interdum eget congue ut, tempor id elit.");
utterance.addEventListener('start', function(e) {
ok(true, 'start utterance 1');
speechSynthesis.cancel();
info('cancel!');
speechSynthesis.speak(utterance2);
info('speak??');
});
var utterance2 = new SpeechSynthesisUtterance("Proin ornare neque vitae " +
"risus mattis rutrum. Suspendisse a velit ut est convallis aliquet." +
"Nullam ante elit, malesuada vel luctus rutrum, ultricies nec libero." +
"Praesent eu iaculis orci. Sed nisl diam, sodales ac purus et," +
"volutpat interdum tortor. Nullam aliquam porta elit et maximus. Cras" +
"risus lectus, elementum vel sodales vel, ultricies eget lectus." +
"Curabitur velit lacus, mollis vel finibus et, molestie sit amet" +
"sapien. Proin vitae dolor ac augue posuere efficitur ac scelerisque" +
"diam. Nulla sed odio elit.");
utterance2.addEventListener('start', function() {
info('start');
speechSynthesis.cancel();
speechSynthesis.speak(utterance3);
});
utterance2.addEventListener('end', function(e) {
gotEndEvent = true;
});
var utterance3 = new SpeechSynthesisUtterance("Hello, world 3!");
utterance3.addEventListener('start', function() {
ok(gotEndEvent, "didn't get start event for this utterance");
});
utterance3.addEventListener('end', done_cb);
// Speak/cancel while paused (Bug 1187105)
speechSynthesis.pause();
speechSynthesis.speak(new SpeechSynthesisUtterance("hello."));
ok(speechSynthesis.pending, "paused speechSynthesis has an utterance queued.");
speechSynthesis.cancel();
ok(!speechSynthesis.pending, "paused speechSynthesis has no utterance queued.");
speechSynthesis.resume();
speechSynthesis.speak(utterance);
ok(!speechSynthesis.speaking, "speechSynthesis is not speaking yet.");
ok(speechSynthesis.pending, "speechSynthesis has an utterance queued.");
}
// Run test with no global queue, and then run it with a global queue.
testFunc(function() {
SpecialPowers.pushPrefEnv(
{ set: [['media.webspeech.synth.force_global_queue', true]] }, function() {
testFunc(SimpleTest.finish)
});
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,46 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=1226015
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 1226015</title>
<script type="application/javascript">
window.SimpleTest = parent.SimpleTest;
window.info = parent.info;
window.is = parent.is;
window.isnot = parent.isnot;
window.ok = parent.ok;
</script>
<script type="application/javascript" src="common.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=1226015">Mozilla Bug 1226015</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="application/javascript">
/** Test for Bug 1226015 **/
function testFunc(done_cb) {
var utterance = new SpeechSynthesisUtterance();
utterance.lang = 'it-IT-failatstart';
speechSynthesis.speak(utterance);
speechSynthesis.cancel();
ok(true, "we didn't crash, that is good.")
SimpleTest.finish();
}
// Run test with no global queue, and then run it with a global queue.
testFunc();
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,85 @@
<!DOCTYPE HTML>
<html lang="en-US">
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=525444
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 525444: Web Speech API, check speech synth queue</title>
<script type="application/javascript">
window.SimpleTest = parent.SimpleTest;
window.is = parent.is;
window.isnot = parent.isnot;
window.ok = parent.ok;
</script>
<script type="application/javascript" src="common.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=525444">Mozilla Bug 525444</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="application/javascript">
/** Test for Bug 525444 **/
// XXX: Rate and pitch are not tested.
var langUriMap = {};
for (var voice of speechSynthesis.getVoices()) {
langUriMap[voice.lang] = voice.voiceURI;
ok(true, voice.lang + ' ' + voice.voiceURI + ' ' + voice.default);
is(voice.default, voice.lang == 'en-JM', 'Only Jamaican voice should be default');
}
ok(langUriMap['en-JM'], 'No English-Jamaican voice');
ok(langUriMap['en-GB'], 'No English-British voice');
ok(langUriMap['en-CA'], 'No English-Canadian voice');
ok(langUriMap['fr-CA'], 'No French-Canadian voice');
ok(langUriMap['es-MX'], 'No Spanish-Mexican voice');
ok(langUriMap['it-IT-fail'], 'No Failing Italian voice');
function testFunc(done_cb) {
synthTestQueue(
[[{text: "Hello, world."},
{ uri: langUriMap['en-JM'] }],
[{text: "Bonjour tout le monde .",
args: { lang: "fr", rate: 0.5, pitch: 0.75 }},
{ uri: langUriMap['fr-CA'], rate: 0.5, pitch: 0.75}],
[{text: "How are you doing?", args: { lang: "en-GB" } },
{ rate: 1, pitch: 1, uri: langUriMap['en-GB']}],
[{text: "Come stai?", args: { lang: "it-IT-fail" } },
{ rate: 1, pitch: 1, uri: langUriMap['it-IT-fail'], err: true }],
[{text: "¡hasta mañana!", args: { lang: "es-MX" } },
{ uri: langUriMap['es-MX'] }]],
function () {
var test_data = [];
var voices = speechSynthesis.getVoices();
for (var voice of voices) {
if (voice.voiceURI.indexOf('urn:moz-tts:fake-direct') < 0) {
continue;
}
test_data.push([{text: "Hello world", args: { voice: voice} },
{uri: voice.voiceURI}]);
}
synthTestQueue(test_data, done_cb);
});
}
// Run test with no global queue, and then run it with a global queue.
testFunc(function() {
SpecialPowers.pushPrefEnv(
{ set: [['media.webspeech.synth.force_global_queue', true]] }, function() {
testFunc(SimpleTest.finish)
});
});
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,53 @@
<!DOCTYPE HTML>
<html>
<!--
https://bugzilla.mozilla.org/show_bug.cgi?id=650295
-->
<head>
<meta charset="utf-8">
<title>Test for Bug 650295: Web Speech API check all classes are present</title>
<script type="application/javascript">
window.SimpleTest = parent.SimpleTest;
window.info = parent.info;
window.is = parent.is;
window.isnot = parent.isnot;
window.ok = parent.ok;
</script>
<script type="application/javascript" src="common.js"></script>
</head>
<body>
<a target="_blank" href="https://bugzilla.mozilla.org/show_bug.cgi?id=650295">Mozilla Bug 650295</a>
<p id="display"></p>
<div id="content" style="display: none">
</div>
<pre id="test">
<script type="application/javascript">
/** Test for Bug 525444 **/
var gotStartEvent = false;
var gotBoundaryEvent = false;
var utterance = new SpeechSynthesisUtterance("Hello, world!");
utterance.addEventListener('start', function(e) {
ok(speechSynthesis.speaking, "speechSynthesis is speaking.");
ok(!speechSynthesis.pending, "speechSynthesis has no other utterances queued.");
gotStartEvent = true;
});
utterance.addEventListener('end', function(e) {
ok(!speechSynthesis.speaking, "speechSynthesis is not speaking.");
ok(!speechSynthesis.pending, "speechSynthesis has no other utterances queued.");
ok(gotStartEvent, "Got 'start' event.");
info('end ' + e.elapsedTime);
SimpleTest.finish();
});
speechSynthesis.speak(utterance);
ok(!speechSynthesis.speaking, "speechSynthesis is not speaking yet.");
ok(speechSynthesis.pending, "speechSynthesis has an utterance queued.");
</script>
</pre>
</body>
</html>

View file

@ -0,0 +1,26 @@
[DEFAULT]
tags=msg
subsuite = media
support-files =
common.js
file_bfcache_frame.html
file_setup.html
file_speech_queue.html
file_speech_simple.html
file_speech_cancel.html
file_speech_error.html
file_indirect_service_events.html
file_global_queue.html
file_global_queue_cancel.html
file_global_queue_pause.html
[test_setup.html]
[test_speech_queue.html]
[test_speech_simple.html]
[test_speech_cancel.html]
[test_speech_error.html]
[test_indirect_service_events.html]
[test_global_queue.html]
[test_global_queue_cancel.html]
[test_global_queue_pause.html]
[test_bfcache.html]

View file

@ -0,0 +1,401 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#include "nsISupports.h"
#include "nsFakeSynthServices.h"
#include "nsPrintfCString.h"
#include "nsIWeakReferenceUtils.h"
#include "SharedBuffer.h"
#include "nsISimpleEnumerator.h"
#include "mozilla/dom/nsSynthVoiceRegistry.h"
#include "mozilla/dom/nsSpeechTask.h"
#include "nsThreadUtils.h"
#include "prenv.h"
#include "mozilla/Preferences.h"
#include "mozilla/DebugOnly.h"
#define CHANNELS 1
#define SAMPLERATE 1600
namespace mozilla {
namespace dom {
StaticRefPtr<nsFakeSynthServices> nsFakeSynthServices::sSingleton;
enum VoiceFlags
{
eSuppressEvents = 1,
eSuppressEnd = 2,
eFailAtStart = 4,
eFail = 8
};
struct VoiceDetails
{
const char* uri;
const char* name;
const char* lang;
bool defaultVoice;
uint32_t flags;
};
static const VoiceDetails sDirectVoices[] = {
{"urn:moz-tts:fake-direct:bob", "Bob Marley", "en-JM", true, 0},
{"urn:moz-tts:fake-direct:amy", "Amy Winehouse", "en-GB", false, 0},
{"urn:moz-tts:fake-direct:lenny", "Leonard Cohen", "en-CA", false, 0},
{"urn:moz-tts:fake-direct:celine", "Celine Dion", "fr-CA", false, 0},
{"urn:moz-tts:fake-direct:julie", "Julieta Venegas", "es-MX", false, },
};
static const VoiceDetails sIndirectVoices[] = {
{"urn:moz-tts:fake-indirect:zanetta", "Zanetta Farussi", "it-IT", false, 0},
{"urn:moz-tts:fake-indirect:margherita", "Margherita Durastanti", "it-IT-noevents-noend", false, eSuppressEvents | eSuppressEnd},
{"urn:moz-tts:fake-indirect:teresa", "Teresa Cornelys", "it-IT-noend", false, eSuppressEnd},
{"urn:moz-tts:fake-indirect:cecilia", "Cecilia Bartoli", "it-IT-failatstart", false, eFailAtStart},
{"urn:moz-tts:fake-indirect:gottardo", "Gottardo Aldighieri", "it-IT-fail", false, eFail},
};
// FakeSynthCallback
class FakeSynthCallback : public nsISpeechTaskCallback
{
public:
explicit FakeSynthCallback(nsISpeechTask* aTask) : mTask(aTask) { }
NS_DECL_CYCLE_COLLECTING_ISUPPORTS
NS_DECL_CYCLE_COLLECTION_CLASS_AMBIGUOUS(FakeSynthCallback, nsISpeechTaskCallback)
NS_IMETHOD OnPause() override
{
if (mTask) {
mTask->DispatchPause(1.5, 1);
}
return NS_OK;
}
NS_IMETHOD OnResume() override
{
if (mTask) {
mTask->DispatchResume(1.5, 1);
}
return NS_OK;
}
NS_IMETHOD OnCancel() override
{
if (mTask) {
mTask->DispatchEnd(1.5, 1);
}
return NS_OK;
}
NS_IMETHOD OnVolumeChanged(float aVolume) override
{
return NS_OK;
}
private:
virtual ~FakeSynthCallback() { }
nsCOMPtr<nsISpeechTask> mTask;
};
NS_IMPL_CYCLE_COLLECTION(FakeSynthCallback, mTask);
NS_INTERFACE_MAP_BEGIN_CYCLE_COLLECTION(FakeSynthCallback)
NS_INTERFACE_MAP_ENTRY(nsISpeechTaskCallback)
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsISpeechTaskCallback)
NS_INTERFACE_MAP_END
NS_IMPL_CYCLE_COLLECTING_ADDREF(FakeSynthCallback)
NS_IMPL_CYCLE_COLLECTING_RELEASE(FakeSynthCallback)
// FakeDirectAudioSynth
class FakeDirectAudioSynth : public nsISpeechService
{
public:
FakeDirectAudioSynth() { }
NS_DECL_ISUPPORTS
NS_DECL_NSISPEECHSERVICE
private:
virtual ~FakeDirectAudioSynth() { }
};
NS_IMPL_ISUPPORTS(FakeDirectAudioSynth, nsISpeechService)
NS_IMETHODIMP
FakeDirectAudioSynth::Speak(const nsAString& aText, const nsAString& aUri,
float aVolume, float aRate, float aPitch,
nsISpeechTask* aTask)
{
class Runnable final : public mozilla::Runnable
{
public:
Runnable(nsISpeechTask* aTask, const nsAString& aText) :
mTask(aTask), mText(aText)
{
}
NS_IMETHOD Run() override
{
RefPtr<FakeSynthCallback> cb = new FakeSynthCallback(nullptr);
mTask->Setup(cb, CHANNELS, SAMPLERATE, 2);
// Just an arbitrary multiplier. Pretend that each character is
// synthesized to 40 frames.
uint32_t frames_length = 40 * mText.Length();
auto frames = MakeUnique<int16_t[]>(frames_length);
mTask->SendAudioNative(frames.get(), frames_length);
mTask->SendAudioNative(nullptr, 0);
return NS_OK;
}
private:
nsCOMPtr<nsISpeechTask> mTask;
nsString mText;
};
nsCOMPtr<nsIRunnable> runnable = new Runnable(aTask, aText);
NS_DispatchToMainThread(runnable);
return NS_OK;
}
NS_IMETHODIMP
FakeDirectAudioSynth::GetServiceType(SpeechServiceType* aServiceType)
{
*aServiceType = nsISpeechService::SERVICETYPE_DIRECT_AUDIO;
return NS_OK;
}
// FakeDirectAudioSynth
class FakeIndirectAudioSynth : public nsISpeechService
{
public:
FakeIndirectAudioSynth() {}
NS_DECL_ISUPPORTS
NS_DECL_NSISPEECHSERVICE
private:
virtual ~FakeIndirectAudioSynth() { }
};
NS_IMPL_ISUPPORTS(FakeIndirectAudioSynth, nsISpeechService)
NS_IMETHODIMP
FakeIndirectAudioSynth::Speak(const nsAString& aText, const nsAString& aUri,
float aVolume, float aRate, float aPitch,
nsISpeechTask* aTask)
{
class DispatchStart final : public Runnable
{
public:
explicit DispatchStart(nsISpeechTask* aTask) :
mTask(aTask)
{
}
NS_IMETHOD Run() override
{
mTask->DispatchStart();
return NS_OK;
}
private:
nsCOMPtr<nsISpeechTask> mTask;
};
class DispatchEnd final : public Runnable
{
public:
DispatchEnd(nsISpeechTask* aTask, const nsAString& aText) :
mTask(aTask), mText(aText)
{
}
NS_IMETHOD Run() override
{
mTask->DispatchEnd(mText.Length()/2, mText.Length());
return NS_OK;
}
private:
nsCOMPtr<nsISpeechTask> mTask;
nsString mText;
};
class DispatchError final : public Runnable
{
public:
DispatchError(nsISpeechTask* aTask, const nsAString& aText) :
mTask(aTask), mText(aText)
{
}
NS_IMETHOD Run() override
{
mTask->DispatchError(mText.Length()/2, mText.Length());
return NS_OK;
}
private:
nsCOMPtr<nsISpeechTask> mTask;
nsString mText;
};
uint32_t flags = 0;
for (uint32_t i = 0; i < ArrayLength(sIndirectVoices); i++) {
if (aUri.EqualsASCII(sIndirectVoices[i].uri)) {
flags = sIndirectVoices[i].flags;
}
}
if (flags & eFailAtStart) {
return NS_ERROR_FAILURE;
}
RefPtr<FakeSynthCallback> cb = new FakeSynthCallback(
(flags & eSuppressEvents) ? nullptr : aTask);
aTask->Setup(cb, 0, 0, 0);
nsCOMPtr<nsIRunnable> runnable = new DispatchStart(aTask);
NS_DispatchToMainThread(runnable);
if (flags & eFail) {
runnable = new DispatchError(aTask, aText);
NS_DispatchToMainThread(runnable);
} else if ((flags & eSuppressEnd) == 0) {
runnable = new DispatchEnd(aTask, aText);
NS_DispatchToMainThread(runnable);
}
return NS_OK;
}
NS_IMETHODIMP
FakeIndirectAudioSynth::GetServiceType(SpeechServiceType* aServiceType)
{
*aServiceType = nsISpeechService::SERVICETYPE_INDIRECT_AUDIO;
return NS_OK;
}
// nsFakeSynthService
NS_INTERFACE_MAP_BEGIN(nsFakeSynthServices)
NS_INTERFACE_MAP_ENTRY(nsIObserver)
NS_INTERFACE_MAP_ENTRY_AMBIGUOUS(nsISupports, nsIObserver)
NS_INTERFACE_MAP_END
NS_IMPL_ADDREF(nsFakeSynthServices)
NS_IMPL_RELEASE(nsFakeSynthServices)
nsFakeSynthServices::nsFakeSynthServices()
{
}
nsFakeSynthServices::~nsFakeSynthServices()
{
}
static void
AddVoices(nsISpeechService* aService, const VoiceDetails* aVoices, uint32_t aLength)
{
RefPtr<nsSynthVoiceRegistry> registry = nsSynthVoiceRegistry::GetInstance();
for (uint32_t i = 0; i < aLength; i++) {
NS_ConvertUTF8toUTF16 name(aVoices[i].name);
NS_ConvertUTF8toUTF16 uri(aVoices[i].uri);
NS_ConvertUTF8toUTF16 lang(aVoices[i].lang);
// These services can handle more than one utterance at a time and have
// several speaking simultaniously. So, aQueuesUtterances == false
registry->AddVoice(aService, uri, name, lang, true, false);
if (aVoices[i].defaultVoice) {
registry->SetDefaultVoice(uri, true);
}
}
registry->NotifyVoicesChanged();
}
void
nsFakeSynthServices::Init()
{
mDirectService = new FakeDirectAudioSynth();
AddVoices(mDirectService, sDirectVoices, ArrayLength(sDirectVoices));
mIndirectService = new FakeIndirectAudioSynth();
AddVoices(mIndirectService, sIndirectVoices, ArrayLength(sIndirectVoices));
}
// nsIObserver
NS_IMETHODIMP
nsFakeSynthServices::Observe(nsISupports* aSubject, const char* aTopic,
const char16_t* aData)
{
MOZ_ASSERT(NS_IsMainThread());
if(NS_WARN_IF(!(!strcmp(aTopic, "speech-synth-started")))) {
return NS_ERROR_UNEXPECTED;
}
if (Preferences::GetBool("media.webspeech.synth.test")) {
NS_DispatchToMainThread(NewRunnableMethod(this, &nsFakeSynthServices::Init));
}
return NS_OK;
}
// static methods
nsFakeSynthServices*
nsFakeSynthServices::GetInstance()
{
MOZ_ASSERT(NS_IsMainThread());
if (!XRE_IsParentProcess()) {
MOZ_ASSERT(false, "nsFakeSynthServices can only be started on main gecko process");
return nullptr;
}
if (!sSingleton) {
sSingleton = new nsFakeSynthServices();
}
return sSingleton;
}
already_AddRefed<nsFakeSynthServices>
nsFakeSynthServices::GetInstanceForService()
{
RefPtr<nsFakeSynthServices> picoService = GetInstance();
return picoService.forget();
}
void
nsFakeSynthServices::Shutdown()
{
if (!sSingleton) {
return;
}
sSingleton = nullptr;
}
} // namespace dom
} // namespace mozilla

View file

@ -0,0 +1,52 @@
/* -*- Mode: C++; tab-width: 2; indent-tabs-mode: nil; c-basic-offset: 2 -*- */
/* vim:set ts=2 sw=2 sts=2 et cindent: */
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
#ifndef nsFakeSynthServices_h
#define nsFakeSynthServices_h
#include "nsTArray.h"
#include "nsIObserver.h"
#include "nsIThread.h"
#include "nsISpeechService.h"
#include "nsRefPtrHashtable.h"
#include "mozilla/StaticPtr.h"
#include "mozilla/Monitor.h"
namespace mozilla {
namespace dom {
class nsFakeSynthServices : public nsIObserver
{
public:
NS_DECL_ISUPPORTS
NS_DECL_NSIOBSERVER
nsFakeSynthServices();
static nsFakeSynthServices* GetInstance();
static already_AddRefed<nsFakeSynthServices> GetInstanceForService();
static void Shutdown();
private:
virtual ~nsFakeSynthServices();
void Init();
nsCOMPtr<nsISpeechService> mDirectService;
nsCOMPtr<nsISpeechService> mIndirectService;
static StaticRefPtr<nsFakeSynthServices> sSingleton;
};
} // namespace dom
} // namespace mozilla
#endif

Some files were not shown because too many files have changed in this diff Show more