82 lines
2.9 KiB
Swift
82 lines
2.9 KiB
Swift
import AVFoundation
|
|||
|
|
import Recording
|
||
|
|
import Speech
|
||
|
|
|
||
|
|
/// The transcribing service used by the Attendi sample app, which transcribes recorded audio into text on device.
|
||
|
|
///
|
||
|
|
/// The service writes the recorded audio into a temporary `.m4a` file and runs it through a `SpeechAnalyzer` with a `SpeechTranscriber`
|
||
|
|
/// module, joining the finalized results into the returned text. The speech model assets for the current locale are downloaded and installed
|
||
|
|
/// on first use; every transcription after that happens entirely offline.
|
||
|
|
struct AudioTranscribingService: TranscribingService {
|
||
|
|
|
||
|
|
// MARK: Initializers
|
||
|
|
|
||
|
|
/// Creates a speech transcribing service.
|
||
|
|
init() {}
|
||
|
|
|
||
|
|
// MARK: Methods
|
||
|
|
|
||
|
|
/// Transcribes the given recorded audio into text, deleting the temporary audio file when the transcription finishes.
|
||
|
|
///
|
||
|
|
/// - Parameter audio: The recorded audio to transcribe.
|
||
|
|
/// - Returns: The transcription of the recorded audio.
|
||
|
|
/// - Throws: ``AudioTranscribingError/localeNotSupported`` when the transcriber does not support the current locale, or any
|
||
|
|
/// error thrown while installing the speech model assets, reading the audio file, or analyzing its contents.
|
||
|
|
func transcribe(_ audio: Data) async throws -> String {
|
||
|
|
try audio.write(to: Constant.File.url)
|
||
|
|
|
||
|
|
defer {
|
||
|
|
try? FileManager.default.removeItem(at: Constant.File.url)
|
||
|
|
}
|
||
|
|
|
||
|
|
guard let locale = await SpeechTranscriber.supportedLocale(equivalentTo: .current) else {
|
||
|
|
throw AudioTranscribingError.localeNotSupported
|
||
|
|
}
|
||
|
|
|
||
|
|
let transcriber = SpeechTranscriber(
|
||
|
|
locale: locale,
|
||
|
|
preset: .transcription
|
||
|
|
)
|
||
|
|
|
||
|
|
if let request = try await AssetInventory.assetInstallationRequest(supporting: [transcriber]) {
|
||
|
|
try await request.downloadAndInstall()
|
||
|
|
}
|
||
|
|
|
||
|
|
let analyzer = SpeechAnalyzer(modules: [transcriber])
|
||
|
|
|
||
|
|
async let text = transcriber.results.reduce(into: "") { text, result in
|
||
|
|
text += String(result.text.characters)
|
||
|
|
}
|
||
|
|
|
||
|
|
let file = try AVAudioFile(forReading: Constant.File.url)
|
||
|
|
|
||
|
|
if let lastSampleTime = try await analyzer.analyzeSequence(from: file) {
|
||
|
|
try await analyzer.finalizeAndFinish(through: lastSampleTime)
|
||
|
|
} else {
|
||
|
|
await analyzer.cancelAndFinishNow()
|
||
|
|
}
|
||
|
|
|
||
|
|
return try await text
|
||
|
|
}
|
||
|
|
|
||
|
|
}
|
||
|
|
|
||
|
|
// MARK: - Errors
|
||
|
|
|
||
|
|
/// The errors thrown by ``AudioTranscribingService``.
|
||
|
|
enum AudioTranscribingError: Error {
|
||
|
|
/// The transcriber does not support the current locale.
|
||
|
|
case localeNotSupported
|
||
|
|
}
|
||
|
|
|
||
|
|
// MARK: - Constants
|
||
|
|
|
||
|
|
/// The constant values used across the speech transcribing service.
|
||
|
|
private nonisolated enum Constant {
|
||
|
|
/// The file constants.
|
||
|
|
enum File {
|
||
|
|
/// The location of the temporary file the recorded audio is written into for the analysis.
|
||
|
|
static let url = FileManager.default.temporaryDirectory.appending(path: "transcription.m4a")
|
||
|
|
}
|
||
|
|
}
|