Compare commits
34 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c4cf27553a | |||
| 087e4b1642 | |||
| 0b1ba5c8cd | |||
| b7455d26d1 | |||
| a311cc583f | |||
| 8b3de5918c | |||
| cf7bc88528 | |||
| 25f017f319 | |||
| 307eae147d | |||
| c9d11316ff | |||
| 71b0c45a5c | |||
| fd82e3d9d9 | |||
| d260c7824d | |||
| b3b11c8649 | |||
| 00985c0d33 | |||
| 1c94f55a9b | |||
| 2040510e9e | |||
| 3449a9bb9c | |||
| ac0ae46cfa | |||
| 8c3ee0f4e7 | |||
| abaa019bd2 | |||
| 98e33b7bcb | |||
| ee8968e2d6 | |||
| 4bc36019c2 | |||
| 1b537eb395 | |||
| 80d7555b60 | |||
| c696fa0197 | |||
| a0902f6eec | |||
| b7ce5e61d8 | |||
| d10f16a119 | |||
| a7be157d06 | |||
| bfce968fb6 | |||
| 95f11b17c7 | |||
| 756c1b3535 |
@@ -7,3 +7,5 @@ DerivedData/
|
||||
.swiftpm/configuration/registries.json
|
||||
.swiftpm/xcode/package.xcworkspace/contents.xcworkspacedata
|
||||
.netrc
|
||||
*.pcm
|
||||
*.wav
|
||||
|
||||
@@ -1,15 +0,0 @@
|
||||
{
|
||||
"originHash" : "5f2b81278809343fed36ed8c17e7d6930bfd5b85261cdf5dadb17ab7ffdfc0e3",
|
||||
"pins" : [
|
||||
{
|
||||
"identity" : "swift-argument-parser",
|
||||
"kind" : "remoteSourceControl",
|
||||
"location" : "https://github.com/apple/swift-argument-parser.git",
|
||||
"state" : {
|
||||
"revision" : "011f0c765fb46d9cac61bca19be0527e99c98c8b",
|
||||
"version" : "1.5.1"
|
||||
}
|
||||
}
|
||||
],
|
||||
"version" : 3
|
||||
}
|
||||
+38
-18
@@ -4,21 +4,41 @@
|
||||
import PackageDescription
|
||||
|
||||
let package = Package(
|
||||
name: "audiotee",
|
||||
platforms: [
|
||||
.macOS("14.2")
|
||||
],
|
||||
dependencies: [
|
||||
.package(url: "https://github.com/apple/swift-argument-parser.git", from: "1.3.0")
|
||||
],
|
||||
targets: [
|
||||
// Targets are the basic building blocks of a package, defining a module or a test suite.
|
||||
// Targets can depend on other targets in this package and products from dependencies.
|
||||
.executableTarget(
|
||||
name: "audiotee",
|
||||
dependencies: [
|
||||
.product(name: "ArgumentParser", package: "swift-argument-parser")
|
||||
]
|
||||
)
|
||||
]
|
||||
)
|
||||
name: "audiotee",
|
||||
platforms: [
|
||||
.macOS("14.2")
|
||||
],
|
||||
products: [
|
||||
// Library that can be imported by other packages
|
||||
.library(
|
||||
name: "AudioTeeCore",
|
||||
targets: ["AudioTeeCore"]
|
||||
),
|
||||
// CLI executable
|
||||
.executable(
|
||||
name: "audiotee",
|
||||
targets: ["AudioTeeCLI"]
|
||||
)
|
||||
],
|
||||
targets: [
|
||||
// Core library with all business logic
|
||||
.target(
|
||||
name: "AudioTeeCore",
|
||||
path: "Sources/AudioTeeCore"
|
||||
),
|
||||
|
||||
// CLI executable that uses the library
|
||||
.executableTarget(
|
||||
name: "AudioTeeCLI",
|
||||
dependencies: ["AudioTeeCore"],
|
||||
path: "Sources/AudioTeeCLI"
|
||||
),
|
||||
|
||||
// Tests for the library
|
||||
.testTarget(
|
||||
name: "AudioTeeCoreTests",
|
||||
dependencies: ["AudioTeeCore"],
|
||||
path: "Tests/AudioTeeCoreTests"
|
||||
)
|
||||
]
|
||||
)
|
||||
@@ -1,12 +1,24 @@
|
||||
# AudioTee
|
||||
|
||||
AudioTee captures your Mac's system audio output and writes PCM encoded chunks of it to `stdout` at regular intervals, either in base64-encoded JSON (good for humans, easy on terminals) or binary (good for other programs). It uses the [Core Audio taps](https://developer.apple.com/documentation/coreaudio/capturing-system-audio-with-core-audio-taps) API introduced in macOS 14.2 (released in December 2023). You can do whatever you want with this audio - stream it somewhere else, save it to disk, visualize it, etc.
|
||||
**⚠️ API Instability Warning: The AudioTee API is unstable at present and subject to change without notice.**
|
||||
|
||||
By default, it taps the audio output from **all** running process and selects the most appropriate audio chunk output format to use based on the presence of a tty. Tap output is forced to `mono` (not configurable) and preserves your output device's sample rate unless you pass a `--sample-rate` flag. Only the default output device is currently supported.
|
||||
AudioTee captures your Mac's system audio output and writes it in PCM encoded chunks to `stdout` at regular intervals. All logging and metadata information is written to `stderr`, meaning at its simplest you can capture whatever's playing through your speakers to a file like this:
|
||||
|
||||
My original (and so far only) use case is streaming audio to a parent process which communicates with a realtime ASR service, so AudioTee makes some design decisions you might not agree with. Open an issue or a PR and we can talk about them. I'm also no Swift developer, so contributions improving codebase idioms and general hygiene are welcome.
|
||||
```bash
|
||||
/path/to/audiotee > output.pcm
|
||||
```
|
||||
|
||||
Recording system audio is harder than it should be on macOS, and folks often wrestle with outdated advice and poorly documented APIs. It's a boring problem which stands in the way of lots of fun applications. There's more code here than you need to solve this problem yourself: the main classes of interest are probably `Core/AudioTapManager` and `Core/AudioRecorder`. Everything's wired together in `CLI/AudioTee`. The rest is just CLI configuration support, output formatting logic, and some utility functions you could probably live without.
|
||||
It's more likely you want to capture this output programmatically. Check out [AudioTee.js](https://github.com/makeusabrew/audioteejs) for a simple Node.js package which does this.
|
||||
|
||||
System audio is captured using the [Core Audio taps](https://developer.apple.com/documentation/coreaudio/capturing-system-audio-with-core-audio-taps) API introduced in macOS 14.2 (released in December 2023). You can do whatever you want with this audio - save it to disk, visualise it, transcribe it, etc.
|
||||
|
||||
By default, AudioTee captures audio output from **all** running processes. Tap output defaults to `mono` (configurable via the `--stereo` flag) and preserves your output device's sample rate (configurable via the `--sample-rate` flag). Only the default output device is currently supported.
|
||||
|
||||
My original (and so far only) use case is streaming audio to a parent process which communicates with a realtime ASR service, so AudioTee makes some design decisions you might not agree with. Open an issue or a PR and we can talk about them. I'm also no Swift developer, so contributions improving codebase idioms and general hygiene are welcome. I have internal variations (and, frankly, improvements) of AudioTee which allow recording mic input as well as system audio, and I'm open to making that part of the main API.
|
||||
|
||||
## Why?
|
||||
|
||||
Recording system audio is harder than it should be on macOS, and folks often wrestle with outdated advice and poorly documented APIs. It's a boring problem which stands in the way of lots of fun applications. There's more code here than you need to solve this problem yourself: the main classes of interest are probably [`Core/AudioTapManager`](https://github.com/makeusabrew/audiotee/blob/main/Sources/Core/AudioTapManager.swift) and [`Core/AudioRecorder`](https://github.com/makeusabrew/audiotee/blob/main/Sources/Core/AudioRecorder.swift). Everything's wired together in [`CLI/AudioTee`](https://github.com/makeusabrew/audiotee/blob/main/Sources/CLI/AudioTee.swift). The rest is just CLI configuration support, output formatting logic, and some utility functions you could probably live without.
|
||||
|
||||
## Requirements
|
||||
|
||||
@@ -16,12 +28,26 @@ Recording system audio is harder than it should be on macOS, and folks often wre
|
||||
|
||||
## Quick start
|
||||
|
||||
The following will start capturing audio output from all running programs and write binary chunks of raw PCM audio data to your terminal:
|
||||
|
||||
```bash
|
||||
git clone git@github.com:makeusabrew/audiotee.git
|
||||
cd audiotee
|
||||
swift run
|
||||
```
|
||||
|
||||
More usefully, you can redirect `stdout` to a file:
|
||||
|
||||
```bash
|
||||
swift run audiotee --sample-rate 16000 > output.pcm
|
||||
```
|
||||
|
||||
Which you can play back using something like `ffplay`:
|
||||
|
||||
```bash
|
||||
ffplay -f s16le -ar 16000 output.pcm
|
||||
```
|
||||
|
||||
## Build
|
||||
|
||||
```bash
|
||||
@@ -36,20 +62,29 @@ swift build -c release
|
||||
Replace the path below with `.build/<arch>/<target>/audiotee`, e.g. `build/arm64-apple-macosx/release/audiotee` for a release build on Apple Silicon.
|
||||
|
||||
```bash
|
||||
# Auto-detect output format (JSON in terminal, binary when piped)
|
||||
# Write raw PCM audio to stdout (logs go to stderr)
|
||||
./audiotee
|
||||
|
||||
# Always use JSON format (terminal-safe)
|
||||
./audiotee --format json
|
||||
# Redirect audio to a file
|
||||
./audiotee > output.pcm
|
||||
|
||||
# Always use binary format (pipe-optimised)
|
||||
./audiotee --format binary
|
||||
# Pipe to another program
|
||||
./audiotee | your_audio_processing_tool
|
||||
|
||||
# Redirect logs as well
|
||||
./audiotee > captured_audio.pcm 2> audiotee.log
|
||||
```
|
||||
|
||||
### Audio conversion
|
||||
|
||||
Note that performing _any_ sample rate conversion will also convert the output bit depth to
|
||||
16-bit - assuming an original depth of 32-bit this results in a loss of dynamic range in exchange for a 50% reduction in output size. For ASR services, 16-bit is sufficient, but it's a non-obvious behaviour worth being aware of.
|
||||
|
||||
```bash
|
||||
# Convert to 16kHz mono (useful for ASR services)
|
||||
# No sample rate preserves your device's default (probably 44.1 or 48kHz with 32-bit float bit depth)
|
||||
./audiotee
|
||||
|
||||
# Any sample rate (even one matching your device default) converts to 16-bit signed integers (half the bandwidth)
|
||||
./audiotee --sample-rate 16000
|
||||
|
||||
# Other supported sample rates: 22050, 24000, 32000, 44100, 48000
|
||||
@@ -60,6 +95,8 @@ Replace the path below with `.build/<arch>/<target>/audiotee`, e.g. `build/arm64
|
||||
|
||||
For now, only a subset of the `CATapDescription` (https://developer.apple.com/documentation/coreaudio/capturing-system-audio-with-core-audio-taps) interface is exposed. PRs welcome.
|
||||
|
||||
Note that trying to include or exclude a PID which isn't currently playing audio will probably fail to convert to an Audio Object and will cause the process to exit.
|
||||
|
||||
```bash
|
||||
# Tap all system audio (default)
|
||||
./audiotee
|
||||
@@ -85,162 +122,51 @@ For now, only a subset of the `CATapDescription` (https://developer.apple.com/do
|
||||
./audiotee --chunk-duration 0.1
|
||||
```
|
||||
|
||||
## Output formats
|
||||
## Output
|
||||
|
||||
AudioTee supports two output formats optimised for different use cases:
|
||||
AudioTee writes raw PCM audio data directly to `stdout` in chunks. All logging, metadata, and status information is written to `stderr`.
|
||||
|
||||
### JSON format (`--format json` or auto in terminal)
|
||||
### Audio format
|
||||
|
||||
JSON messages to stdout, one per line. Audio data is base64-encoded for terminal safety.
|
||||
- **Format**: Raw PCM audio data
|
||||
- **Channels**: 1 in Mono mode (default), 2 in stereo mode
|
||||
- **Sample rate**: Matches your output device's sample rate by default (configurable)
|
||||
- **Bit depth**: 32-bit float by default, or 16-bit when sample rate conversion is performed
|
||||
- **Endianness**: Little-endian
|
||||
- **Chunk duration**: 200ms by default (configurable)
|
||||
|
||||
### Binary format (`--format binary` or auto when piped)
|
||||
### Logs and monitoring
|
||||
|
||||
JSON metadata lines followed by raw binary audio data. More efficient for piping to other processes.
|
||||
All program logs are written to `stderr` and can be captured separately:
|
||||
|
||||
## Protocol
|
||||
```bash
|
||||
# Capture audio and logs separately
|
||||
./audiotee > audio.pcm 2> audiotee.log
|
||||
|
||||
### Message types
|
||||
|
||||
All messages (except raw binary audio chunks) follow this envelope structure:
|
||||
|
||||
```json
|
||||
{
|
||||
"timestamp": "2024-03-21T15:30:45.123Z",
|
||||
"message_type": "...",
|
||||
"data": { ... }
|
||||
}
|
||||
# View logs in real-time while capturing audio
|
||||
./audiotee > audio.pcm 2>&1 | grep "AudioTee"
|
||||
```
|
||||
|
||||
#### 1. Metadata
|
||||
|
||||
Sent once at startup to describe the audio format:
|
||||
|
||||
```json
|
||||
{
|
||||
"timestamp": "2024-03-21T15:30:45.123Z",
|
||||
"message_type": "metadata",
|
||||
"data": {
|
||||
"sample_rate": 48000,
|
||||
"channels_per_frame": 1,
|
||||
"bits_per_channel": 32,
|
||||
"is_float": true,
|
||||
"capture_mode": "audio",
|
||||
"device_name": null,
|
||||
"device_uid": null,
|
||||
"encoding": "pcm_f32le"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
#### 2. Stream start
|
||||
|
||||
Indicates audio data will follow:
|
||||
|
||||
```json
|
||||
{
|
||||
"timestamp": "2024-03-21T15:30:45.123Z",
|
||||
"message_type": "stream_start",
|
||||
"data": null
|
||||
}
|
||||
```
|
||||
|
||||
#### 3. Audio data
|
||||
|
||||
**JSON format:**
|
||||
|
||||
```json
|
||||
{
|
||||
"timestamp": "2024-03-21T15:30:45.123Z",
|
||||
"message_type": "audio",
|
||||
"data": {
|
||||
"timestamp": "2024-03-21T15:30:45.123Z",
|
||||
"duration": 0.2,
|
||||
"peak_amplitude": 0.45,
|
||||
"audio_data": "base64_encoded_raw_audio..."
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
**Binary format:**
|
||||
|
||||
```json
|
||||
{
|
||||
"timestamp": "2024-03-21T15:30:45.123Z",
|
||||
"message_type": "audio",
|
||||
"data": {
|
||||
"timestamp": "2024-03-21T15:30:45.123Z",
|
||||
"duration": 0.2,
|
||||
"peak_amplitude": 0.45,
|
||||
"audio_length": 9600
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
_Followed immediately by 9600 bytes of raw binary audio data_
|
||||
|
||||
#### 4. Stream stop
|
||||
|
||||
Sent when recording stops:
|
||||
|
||||
```json
|
||||
{
|
||||
"timestamp": "2024-03-21T15:30:45.123Z",
|
||||
"message_type": "stream_stop",
|
||||
"data": null
|
||||
}
|
||||
```
|
||||
|
||||
#### 5. Log messages
|
||||
|
||||
Info, error, and debug messages (useful for monitoring):
|
||||
|
||||
```json
|
||||
{
|
||||
"timestamp": "2024-03-21T15:30:45.123Z",
|
||||
"message_type": "info",
|
||||
"data": {
|
||||
"message": "Starting AudioTee...",
|
||||
"context": { "output_format": "auto" }
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Consuming output
|
||||
|
||||
**JSON format:**
|
||||
|
||||
1. Parse each line as JSON using the envelope structure
|
||||
2. Use `metadata` message to understand the audio format
|
||||
3. For `audio` messages, decode `audio_data` from base64 to get raw PCM data
|
||||
4. Do something with each chunk of data
|
||||
|
||||
**Binary format:**
|
||||
|
||||
1. Parse JSON metadata lines using the envelope structure
|
||||
2. Use `metadata` message to understand the audio format
|
||||
3. For `audio` messages, read `audio_length` bytes of raw binary data after the JSON line
|
||||
4. Do something with each chunk of data
|
||||
|
||||
**Note**: binary is actually a mixed mode; JSON during boot, JSON packet header information preceding each binary chunk.
|
||||
|
||||
## Command Line options
|
||||
|
||||
- `--format, -f`: Output format (`json`, `binary`, `auto`) [default: `auto`]
|
||||
- `--include-processes`: Process IDs to tap (space-separated, empty = all processes)
|
||||
- `--exclude-processes`: Process IDs to exclude (space-separated, empty = none)
|
||||
- `--mute`: Mute processes being tapped
|
||||
- `--stereo`: Record in stereo
|
||||
- `--sample-rate`: Target sample rate (8000, 16000, 22050, 24000, 32000, 44100, 48000)
|
||||
- `--chunk-duration`: Audio chunk duration in seconds [default: 0.2, max: 5.0]
|
||||
|
||||
## Permissions
|
||||
|
||||
There is no provision in the code to pre-emptively check for the required `NSAudioCaptureUsageDescription` permission,
|
||||
so you'll be prompted the first time AudioTee tries to record anything. If you want to check and/or request permissions ahead of time, check out [AudioCap's clever TCC probing approach](https://github.com/insidegui/AudioCap/blob/main/AudioCap/ProcessTap/AudioRecordingPermission.swift).
|
||||
There is no provision in the code to pre-emptively check for the required `NSAudioCaptureUsageDescription` permission, so you'll be prompted the first time AudioTee tries to record anything. Note that some terminal emulators like iTerm don't always prompt for these permissions (though the macOS builtin terminal definitely does), so you might need to grant them ahead of time if audiotee runs but never records anything.
|
||||
|
||||
## References
|
||||
If you want to check and/or request permissions ahead of time, check out [AudioCap's fantastic TCC probing approach](https://github.com/insidegui/AudioCap/blob/main/AudioCap/ProcessTap/AudioRecordingPermission.swift).
|
||||
|
||||
## References / useful links
|
||||
|
||||
- [Apple Core Audio Taps Documentation](https://developer.apple.com/documentation/coreaudio/capturing-system-audio-with-core-audio-taps)
|
||||
- [AudioCap Implementation](https://github.com/insidegui/AudioCap)
|
||||
- [AudioTee.js](https://github.com/makeusabrew/audioteejs)
|
||||
|
||||
## License
|
||||
|
||||
|
||||
@@ -0,0 +1,229 @@
|
||||
import Foundation
|
||||
|
||||
// MARK: - Error Types
|
||||
|
||||
enum ArgumentParserError: Error, CustomStringConvertible {
|
||||
case unknownOption(String)
|
||||
case missingValue(String)
|
||||
case invalidValue(String, String)
|
||||
case validationFailed(String)
|
||||
case helpRequested
|
||||
|
||||
var description: String {
|
||||
switch self {
|
||||
case .unknownOption(let option):
|
||||
return "Unknown option: \(option)"
|
||||
case .missingValue(let option):
|
||||
return "Missing value for option: \(option)"
|
||||
case .invalidValue(let option, let value):
|
||||
return "Invalid value '\(value)' for option: \(option)"
|
||||
case .validationFailed(let message):
|
||||
return message
|
||||
case .helpRequested:
|
||||
return "" // Help is handled separately
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Argument Configuration
|
||||
|
||||
struct ArgumentConfig {
|
||||
let name: String
|
||||
let shortName: String?
|
||||
let help: String
|
||||
let isFlag: Bool
|
||||
let isArray: Bool
|
||||
let defaultValue: String?
|
||||
|
||||
init(
|
||||
name: String, shortName: String? = nil, help: String, isFlag: Bool = false,
|
||||
isArray: Bool = false, defaultValue: String? = nil
|
||||
) {
|
||||
self.name = name
|
||||
self.shortName = shortName
|
||||
self.help = help
|
||||
self.isFlag = isFlag
|
||||
self.isArray = isArray
|
||||
self.defaultValue = defaultValue
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Simple Argument Parser
|
||||
|
||||
class SimpleArgumentParser {
|
||||
private let programName: String
|
||||
private let abstract: String
|
||||
private let discussion: String
|
||||
private var configs: [ArgumentConfig] = []
|
||||
private var parsedValues: [String: [String]] = [:]
|
||||
|
||||
init(programName: String, abstract: String, discussion: String = "") {
|
||||
self.programName = programName
|
||||
self.abstract = abstract
|
||||
self.discussion = discussion
|
||||
}
|
||||
|
||||
func addOption(name: String, shortName: String? = nil, help: String, defaultValue: String? = nil)
|
||||
{
|
||||
configs.append(
|
||||
ArgumentConfig(name: name, shortName: shortName, help: help, defaultValue: defaultValue))
|
||||
}
|
||||
|
||||
func addArrayOption(name: String, shortName: String? = nil, help: String) {
|
||||
configs.append(ArgumentConfig(name: name, shortName: shortName, help: help, isArray: true))
|
||||
}
|
||||
|
||||
func addFlag(name: String, shortName: String? = nil, help: String) {
|
||||
configs.append(ArgumentConfig(name: name, shortName: shortName, help: help, isFlag: true))
|
||||
}
|
||||
|
||||
func parse(_ arguments: [String] = Array(CommandLine.arguments.dropFirst())) throws {
|
||||
var i = 0
|
||||
|
||||
while i < arguments.count {
|
||||
let arg = arguments[i]
|
||||
|
||||
if arg == "--help" || arg == "-h" {
|
||||
throw ArgumentParserError.helpRequested
|
||||
}
|
||||
|
||||
guard arg.hasPrefix("-") else {
|
||||
throw ArgumentParserError.unknownOption(arg)
|
||||
}
|
||||
|
||||
let optionName = findOptionName(arg)
|
||||
guard let config = findConfig(optionName) else {
|
||||
throw ArgumentParserError.unknownOption(arg)
|
||||
}
|
||||
|
||||
if config.isFlag {
|
||||
parsedValues[config.name] = ["true"]
|
||||
i += 1
|
||||
} else {
|
||||
// Need a value
|
||||
i += 1
|
||||
guard i < arguments.count else {
|
||||
throw ArgumentParserError.missingValue(arg)
|
||||
}
|
||||
|
||||
if config.isArray {
|
||||
// Collect all values until next option or end
|
||||
var values: [String] = []
|
||||
while i < arguments.count && !arguments[i].hasPrefix("-") {
|
||||
values.append(arguments[i])
|
||||
i += 1
|
||||
}
|
||||
if values.isEmpty {
|
||||
throw ArgumentParserError.missingValue(arg)
|
||||
}
|
||||
parsedValues[config.name] = values
|
||||
} else {
|
||||
let value = arguments[i]
|
||||
parsedValues[config.name] = [value]
|
||||
i += 1
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Set default values for missing options
|
||||
for config in configs {
|
||||
if parsedValues[config.name] == nil, let defaultValue = config.defaultValue {
|
||||
parsedValues[config.name] = [defaultValue]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func findOptionName(_ arg: String) -> String {
|
||||
if arg.hasPrefix("--") {
|
||||
return String(arg.dropFirst(2))
|
||||
} else if arg.hasPrefix("-") {
|
||||
return String(arg.dropFirst(1))
|
||||
}
|
||||
return arg
|
||||
}
|
||||
|
||||
private func findConfig(_ optionName: String) -> ArgumentConfig? {
|
||||
return configs.first { config in
|
||||
config.name == optionName || config.shortName == optionName
|
||||
}
|
||||
}
|
||||
|
||||
func getValue<T>(_ name: String, as type: T.Type) throws -> T {
|
||||
guard let values = parsedValues[name], let value = values.first else {
|
||||
throw ArgumentParserError.missingValue(name)
|
||||
}
|
||||
|
||||
return try convertValue(value, to: type, optionName: name)
|
||||
}
|
||||
|
||||
func getOptionalValue<T>(_ name: String, as type: T.Type) throws -> T? {
|
||||
guard let values = parsedValues[name], let value = values.first else {
|
||||
return nil
|
||||
}
|
||||
|
||||
return try convertValue(value, to: type, optionName: name)
|
||||
}
|
||||
|
||||
func getArrayValue<T>(_ name: String, as type: T.Type) throws -> [T] {
|
||||
guard let values = parsedValues[name] else {
|
||||
return []
|
||||
}
|
||||
|
||||
return try values.map { try convertValue($0, to: type, optionName: name) }
|
||||
}
|
||||
|
||||
func getFlag(_ name: String) -> Bool {
|
||||
return parsedValues[name]?.first == "true"
|
||||
}
|
||||
|
||||
private func convertValue<T>(_ value: String, to type: T.Type, optionName: String) throws -> T {
|
||||
if type == String.self {
|
||||
return value as! T
|
||||
} else if type == Int32.self {
|
||||
guard let intValue = Int32(value) else {
|
||||
throw ArgumentParserError.invalidValue(optionName, value)
|
||||
}
|
||||
return intValue as! T
|
||||
} else if type == Double.self {
|
||||
guard let doubleValue = Double(value) else {
|
||||
throw ArgumentParserError.invalidValue(optionName, value)
|
||||
}
|
||||
return doubleValue as! T
|
||||
}
|
||||
|
||||
throw ArgumentParserError.invalidValue(optionName, value)
|
||||
}
|
||||
|
||||
func printHelp() {
|
||||
print(abstract)
|
||||
|
||||
if !discussion.isEmpty {
|
||||
print("\n\(discussion)")
|
||||
}
|
||||
|
||||
print("\nUSAGE:")
|
||||
print(" \(programName) [OPTIONS]")
|
||||
|
||||
let optionConfigs = configs.filter { !$0.isFlag }
|
||||
let flagConfigs = configs.filter { $0.isFlag }
|
||||
|
||||
if !optionConfigs.isEmpty {
|
||||
print("\nOPTIONS:")
|
||||
for config in optionConfigs {
|
||||
let shortName = config.shortName.map { "-\($0), " } ?? ""
|
||||
let defaultDesc = config.defaultValue.map { " (default: \($0))" } ?? ""
|
||||
print(" \(shortName)--\(config.name) \(config.help)\(defaultDesc)")
|
||||
}
|
||||
}
|
||||
|
||||
if !flagConfigs.isEmpty {
|
||||
print("\nFLAGS:")
|
||||
for config in flagConfigs {
|
||||
let shortName = config.shortName.map { "-\($0), " } ?? ""
|
||||
print(" \(shortName)--\(config.name) \(config.help)")
|
||||
}
|
||||
}
|
||||
|
||||
print("\n -h, --help Show this help message")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,209 @@
|
||||
import AudioTeeCore
|
||||
import CoreAudio
|
||||
import Foundation
|
||||
|
||||
struct AudioTee {
|
||||
var includeProcesses: [Int32] = []
|
||||
var excludeProcesses: [Int32] = []
|
||||
var mute: Bool = false
|
||||
var stereo: Bool = false
|
||||
var sampleRate: Double?
|
||||
var chunkDuration: Double = 0.2
|
||||
var flush: Bool = false
|
||||
|
||||
init() {}
|
||||
|
||||
static func main() {
|
||||
let parser = SimpleArgumentParser(
|
||||
programName: "audiotee",
|
||||
abstract: "Capture system audio and stream to stdout",
|
||||
discussion: """
|
||||
AudioTee captures system audio using Core Audio taps and streams it as structured output.
|
||||
|
||||
Process filtering:
|
||||
• include-processes: Only tap specified process IDs (empty = all processes)
|
||||
• exclude-processes: Tap all processes except specified ones
|
||||
• mute: How to handle processes being tapped
|
||||
|
||||
Examples:
|
||||
audiotee # Auto format, tap all processes
|
||||
audiotee --sample-rate 16000 # Convert to 16kHz mono for ASR
|
||||
audiotee --sample-rate 8000 # Convert to 8kHz for telephony
|
||||
audiotee --include-processes 1234 # Only tap process 1234
|
||||
audiotee --include-processes 1234 5678 9012 # Tap only these processes
|
||||
audiotee --exclude-processes 1234 5678 # Tap everything except these
|
||||
audiotee --mute # Mute processes being tapped
|
||||
audiotee --flush # Flush stdout after each chunk
|
||||
"""
|
||||
)
|
||||
|
||||
// Configure arguments
|
||||
parser.addArrayOption(
|
||||
name: "include-processes",
|
||||
help: "Process IDs to include (space-separated, empty = all processes)")
|
||||
parser.addArrayOption(
|
||||
name: "exclude-processes", help: "Process IDs to exclude (space-separated)")
|
||||
parser.addFlag(name: "mute", help: "Mute processes being tapped")
|
||||
parser.addFlag(name: "stereo", help: "Records in stereo")
|
||||
parser.addFlag(name: "flush", help: "Flush stdout after each audio chunk (reduces latency when piping)")
|
||||
parser.addOption(
|
||||
name: "sample-rate",
|
||||
help: "Target sample rate (8000, 16000, 22050, 24000, 32000, 44100, 48000)")
|
||||
parser.addOption(
|
||||
name: "chunk-duration", help: "Audio chunk duration in seconds", defaultValue: "0.2")
|
||||
|
||||
// Parse arguments
|
||||
do {
|
||||
try parser.parse()
|
||||
|
||||
var audioTee = AudioTee()
|
||||
|
||||
// Extract values
|
||||
audioTee.includeProcesses = try parser.getArrayValue("include-processes", as: Int32.self)
|
||||
audioTee.excludeProcesses = try parser.getArrayValue("exclude-processes", as: Int32.self)
|
||||
audioTee.mute = parser.getFlag("mute")
|
||||
audioTee.stereo = parser.getFlag("stereo")
|
||||
audioTee.flush = parser.getFlag("flush")
|
||||
audioTee.sampleRate = try parser.getOptionalValue("sample-rate", as: Double.self)
|
||||
audioTee.chunkDuration = try parser.getValue("chunk-duration", as: Double.self)
|
||||
|
||||
// Validate
|
||||
try audioTee.validate()
|
||||
|
||||
// Run
|
||||
try audioTee.run()
|
||||
|
||||
} catch ArgumentParserError.helpRequested {
|
||||
parser.printHelp()
|
||||
exit(0)
|
||||
} catch ArgumentParserError.validationFailed(let message) {
|
||||
print("Error: \(message)", to: &standardError)
|
||||
exit(1)
|
||||
} catch let error as ArgumentParserError {
|
||||
print("Error: \(error.description)", to: &standardError)
|
||||
parser.printHelp()
|
||||
exit(1)
|
||||
} catch {
|
||||
print("Error: \(error)", to: &standardError)
|
||||
exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func validate() throws {
|
||||
if !includeProcesses.isEmpty && !excludeProcesses.isEmpty {
|
||||
throw ArgumentParserError.validationFailed(
|
||||
"Cannot specify both --include-processes and --exclude-processes")
|
||||
}
|
||||
}
|
||||
|
||||
func run() throws {
|
||||
setupSignalHandlers()
|
||||
|
||||
Logger.info("Starting AudioTee...")
|
||||
|
||||
// Validate chunk duration
|
||||
guard chunkDuration > 0 && chunkDuration <= 5.0 else {
|
||||
Logger.error(
|
||||
"Invalid chunk duration",
|
||||
context: ["chunk_duration": String(chunkDuration), "valid_range": "0.0 < duration <= 5.0"])
|
||||
throw ExitCode.failure
|
||||
}
|
||||
|
||||
// Convert include/exclude processes to TapConfiguration format
|
||||
let (processes, isExclusive) = convertProcessFlags()
|
||||
|
||||
let tapConfig = TapConfiguration(
|
||||
processes: processes,
|
||||
muteBehavior: mute ? .muted : .unmuted,
|
||||
isExclusive: isExclusive,
|
||||
isMono: !stereo
|
||||
)
|
||||
|
||||
let audioTapManager = AudioTapManager()
|
||||
do {
|
||||
try audioTapManager.setupAudioTap(with: tapConfig)
|
||||
} catch AudioTeeError.pidTranslationFailed(let failedPIDs) {
|
||||
Logger.error(
|
||||
"Failed to translate process IDs to audio objects",
|
||||
context: [
|
||||
"failed_pids": failedPIDs.map(String.init).joined(separator: ", "),
|
||||
"suggestion": "Check that the process IDs exist and are running",
|
||||
])
|
||||
throw ExitCode.failure
|
||||
} catch {
|
||||
Logger.error(
|
||||
"Failed to setup audio tap", context: ["error": String(describing: error)])
|
||||
throw ExitCode.failure
|
||||
}
|
||||
|
||||
guard let deviceID = audioTapManager.getDeviceID() else {
|
||||
Logger.error("Failed to get device ID from audio tap manager")
|
||||
throw ExitCode.failure
|
||||
}
|
||||
|
||||
let outputHandler = BinaryAudioOutputHandler(flushAfterWrite: flush)
|
||||
let recorder = AudioRecorder(
|
||||
deviceID: deviceID, outputHandler: outputHandler, convertToSampleRate: sampleRate,
|
||||
chunkDuration: chunkDuration)
|
||||
recorder.startRecording()
|
||||
|
||||
// Run until the run loop is stopped (by signal handler)
|
||||
while true {
|
||||
let result = CFRunLoopRunInMode(CFRunLoopMode.defaultMode, 0.1, false)
|
||||
if result == CFRunLoopRunResult.stopped || result == CFRunLoopRunResult.finished {
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
Logger.info("Shutting down...")
|
||||
recorder.stopRecording()
|
||||
}
|
||||
|
||||
private func setupSignalHandlers() {
|
||||
signal(SIGINT) { _ in
|
||||
Logger.info("Received SIGINT, initiating graceful shutdown...")
|
||||
CFRunLoopStop(CFRunLoopGetMain())
|
||||
}
|
||||
signal(SIGTERM) { _ in
|
||||
Logger.info("Received SIGTERM, initiating graceful shutdown...")
|
||||
CFRunLoopStop(CFRunLoopGetMain())
|
||||
}
|
||||
}
|
||||
|
||||
private func convertProcessFlags() -> ([Int32], Bool) {
|
||||
if !includeProcesses.isEmpty {
|
||||
// Include specific processes only
|
||||
return (includeProcesses, false)
|
||||
} else if !excludeProcesses.isEmpty {
|
||||
// Exclude specific processes (tap everything except these)
|
||||
return (excludeProcesses, true)
|
||||
} else {
|
||||
// Default: tap everything
|
||||
return ([], true)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Helper for stderr output
|
||||
var standardError = FileHandle.standardError
|
||||
|
||||
extension FileHandle: TextOutputStream {
|
||||
public func write(_ string: String) {
|
||||
let data = Data(string.utf8)
|
||||
self.write(data)
|
||||
}
|
||||
}
|
||||
|
||||
// Exit code handling
|
||||
enum ExitCode: Error {
|
||||
case failure
|
||||
}
|
||||
|
||||
extension ExitCode {
|
||||
var code: Int32 {
|
||||
switch self {
|
||||
case .failure:
|
||||
return 1
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
import ArgumentParser
|
||||
import AudioTeeCore
|
||||
import AudioToolbox
|
||||
import Foundation
|
||||
|
||||
AudioTee.main()
|
||||
AudioTee.main()
|
||||
@@ -0,0 +1,105 @@
|
||||
import CoreAudio
|
||||
import Foundation
|
||||
|
||||
public class AudioBuffer {
|
||||
private var buffer: [UInt8]
|
||||
private var writeIndex: Int = 0
|
||||
private var readIndex: Int = 0
|
||||
private var availableBytes: Int = 0
|
||||
private let maxBufferSize: Int
|
||||
|
||||
private let bytesPerChunk: Int
|
||||
private let chunkDuration: Double
|
||||
|
||||
public init(format: AudioStreamBasicDescription, chunkDuration: Double = 0.2) {
|
||||
|
||||
// Pre-calculate chunk parameters
|
||||
let bytesPerFrame = Int(format.mBytesPerFrame)
|
||||
let samplesPerChunk = Int(format.mSampleRate * chunkDuration)
|
||||
self.bytesPerChunk = samplesPerChunk * bytesPerFrame
|
||||
self.chunkDuration = Double(samplesPerChunk) / format.mSampleRate
|
||||
|
||||
// Calculate max buffer size to hold ~10 seconds of audio, way more than the maximum we allow
|
||||
let bytesPerSecond = Int(format.mSampleRate) * bytesPerFrame
|
||||
self.maxBufferSize = bytesPerSecond * 10
|
||||
|
||||
// Pre-allocated ring buffer
|
||||
self.buffer = Array(repeating: 0, count: maxBufferSize)
|
||||
}
|
||||
|
||||
public func append(_ data: Data) {
|
||||
guard availableBytes + data.count <= maxBufferSize else {
|
||||
Logger.error(
|
||||
"Audio buffer overflow",
|
||||
context: [
|
||||
"requested": String(data.count),
|
||||
"available": String(maxBufferSize - availableBytes),
|
||||
])
|
||||
return
|
||||
}
|
||||
|
||||
data.withUnsafeBytes { bytes in
|
||||
let sourceBytes = bytes.bindMemory(to: UInt8.self)
|
||||
let dataSize = sourceBytes.count
|
||||
|
||||
// Check if we can copy in one block (no wrap-around)
|
||||
if writeIndex + dataSize <= maxBufferSize {
|
||||
// only one write needed
|
||||
buffer.replaceSubrange(writeIndex..<writeIndex + dataSize, with: sourceBytes)
|
||||
writeIndex = (writeIndex + dataSize) % maxBufferSize
|
||||
} else {
|
||||
// two writes needed due to wrap-around
|
||||
let firstChunkSize = maxBufferSize - writeIndex
|
||||
let secondChunkSize = dataSize - firstChunkSize
|
||||
|
||||
buffer.replaceSubrange(writeIndex..<maxBufferSize, with: sourceBytes.prefix(firstChunkSize))
|
||||
buffer.replaceSubrange(0..<secondChunkSize, with: sourceBytes.suffix(secondChunkSize))
|
||||
|
||||
writeIndex = secondChunkSize
|
||||
}
|
||||
}
|
||||
|
||||
availableBytes += data.count
|
||||
}
|
||||
|
||||
public func processChunks() -> [AudioPacket] {
|
||||
var packets: [AudioPacket] = []
|
||||
|
||||
while let packet = nextChunk() {
|
||||
packets.append(packet)
|
||||
}
|
||||
|
||||
return packets
|
||||
}
|
||||
|
||||
private func nextChunk() -> AudioPacket? {
|
||||
// Check if we have enough data for a complete chunk
|
||||
guard availableBytes >= bytesPerChunk else { return nil }
|
||||
|
||||
var chunkData = Data(capacity: bytesPerChunk)
|
||||
|
||||
// Check if we can copy in one block (no wrap-around)
|
||||
if readIndex + bytesPerChunk <= maxBufferSize {
|
||||
// one copy needed
|
||||
chunkData.append(contentsOf: buffer[readIndex..<readIndex + bytesPerChunk])
|
||||
readIndex = (readIndex + bytesPerChunk) % maxBufferSize
|
||||
} else {
|
||||
// two copies needed due to wrap-around
|
||||
let firstChunkSize = maxBufferSize - readIndex
|
||||
let secondChunkSize = bytesPerChunk - firstChunkSize
|
||||
|
||||
chunkData.append(contentsOf: buffer[readIndex..<maxBufferSize])
|
||||
chunkData.append(contentsOf: buffer[0..<secondChunkSize])
|
||||
|
||||
readIndex = secondChunkSize
|
||||
}
|
||||
|
||||
availableBytes -= bytesPerChunk
|
||||
|
||||
return AudioPacket(
|
||||
timestamp: Date(),
|
||||
duration: chunkDuration,
|
||||
data: chunkData
|
||||
)
|
||||
}
|
||||
}
|
||||
+7
-28
@@ -54,12 +54,7 @@ public class AudioFormatConverter {
|
||||
}
|
||||
|
||||
public func transform(_ packet: AudioPacket) -> AudioPacket {
|
||||
let inputData = packet.rawAudioData
|
||||
|
||||
// Short-circuit if no conversion needed
|
||||
if sourceFormat.sampleRate == targetFormat.sampleRate {
|
||||
return packet
|
||||
}
|
||||
let inputData = packet.data
|
||||
|
||||
// Calculate frame counts
|
||||
let inputFrameCount =
|
||||
@@ -124,17 +119,10 @@ public class AudioFormatConverter {
|
||||
return AudioPacket(
|
||||
timestamp: packet.timestamp,
|
||||
duration: packet.duration,
|
||||
peakAmplitude: packet.peakAmplitude,
|
||||
rawAudioData: outputData
|
||||
data: outputData
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Convenience Constructors
|
||||
|
||||
extension AudioFormatConverter {
|
||||
/// Create a converter to a specific sample rate with mono PCM 16-bit output
|
||||
/// Since the tap already converts to mono, we hardcode channels to 1
|
||||
public static func toSampleRate(
|
||||
_ sampleRate: Double, from sourceFormat: AudioStreamBasicDescription
|
||||
) throws -> AudioFormatConverter {
|
||||
@@ -142,26 +130,17 @@ extension AudioFormatConverter {
|
||||
targetFormat.mSampleRate = sampleRate
|
||||
targetFormat.mFormatID = kAudioFormatLinearPCM
|
||||
targetFormat.mFormatFlags = kAudioFormatFlagIsPacked | kAudioFormatFlagIsSignedInteger
|
||||
targetFormat.mBytesPerPacket = 2
|
||||
targetFormat.mFramesPerPacket = 1
|
||||
targetFormat.mBytesPerFrame = 2
|
||||
targetFormat.mChannelsPerFrame = 1 // Always mono since tap handles this
|
||||
targetFormat.mBitsPerChannel = 16
|
||||
targetFormat.mChannelsPerFrame = sourceFormat.mChannelsPerFrame
|
||||
targetFormat.mBytesPerFrame =
|
||||
(targetFormat.mBitsPerChannel / 8) * sourceFormat.mChannelsPerFrame
|
||||
targetFormat.mBytesPerPacket = targetFormat.mFramesPerPacket * targetFormat.mBytesPerFrame
|
||||
|
||||
return try AudioFormatConverter(sourceFormat: sourceFormat, targetFormat: targetFormat)
|
||||
}
|
||||
|
||||
/// Common sample rates for validation
|
||||
public static let supportedSampleRates: [Double] = [
|
||||
8000, 16000, 22050, 24000, 32000, 44100, 48000,
|
||||
]
|
||||
|
||||
/// Validate if a sample rate is supported
|
||||
public static func isValidSampleRate(_ sampleRate: Double) -> Bool {
|
||||
return supportedSampleRates.contains(sampleRate)
|
||||
return [8000, 16000, 22050, 24000, 32000, 44100, 48000].contains(sampleRate)
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Error Types
|
||||
|
||||
// AudioConverterError moved to Sources/Core/Errors/AudioTeeErrors.swift
|
||||
@@ -0,0 +1,116 @@
|
||||
import AudioToolbox
|
||||
import CoreAudio
|
||||
import Foundation
|
||||
|
||||
public class AudioFormatManager {
|
||||
public static func getDeviceFormat(deviceID: AudioObjectID) -> AudioStreamBasicDescription {
|
||||
// First, wait for the device to become alive/ready
|
||||
let deviceReadyTimeout = 2.0 // 2 seconds max wait
|
||||
let pollInterval = 0.1 // 100ms poll interval
|
||||
let maxPolls = Int(deviceReadyTimeout / pollInterval)
|
||||
|
||||
Logger.debug(
|
||||
"Waiting for audio device to become ready", context: ["device_id": String(deviceID)])
|
||||
|
||||
// Poll device readiness
|
||||
for poll in 1...maxPolls {
|
||||
if isAudioDeviceValid(deviceID) {
|
||||
Logger.debug(
|
||||
"Audio device is ready", context: ["device_id": String(deviceID), "polls": String(poll)])
|
||||
break
|
||||
}
|
||||
|
||||
if poll == maxPolls {
|
||||
Logger.info(
|
||||
"Device did not become ready within timeout, proceeding anyway",
|
||||
context: [
|
||||
"device_id": String(deviceID),
|
||||
"timeout_seconds": String(deviceReadyTimeout),
|
||||
])
|
||||
break
|
||||
}
|
||||
|
||||
Logger.info("------- not ready; retrying...")
|
||||
|
||||
Thread.sleep(forTimeInterval: pollInterval)
|
||||
}
|
||||
|
||||
// Now attempt to get the stream format with limited retries
|
||||
let maxRetries = 3 // Reduced since device should be ready
|
||||
let retryDelayMs = 20 // Shorter delay since we've already waited for readiness
|
||||
|
||||
for attempt in 1...maxRetries {
|
||||
var propertyAddress = getPropertyAddress(
|
||||
selector: kAudioDevicePropertyStreamFormat,
|
||||
scope: kAudioDevicePropertyScopeInput)
|
||||
var propertySize = UInt32(MemoryLayout<AudioStreamBasicDescription>.stride)
|
||||
var streamFormat = AudioStreamBasicDescription()
|
||||
let status = AudioObjectGetPropertyData(
|
||||
deviceID, &propertyAddress, 0, nil, &propertySize, &streamFormat)
|
||||
|
||||
if status == noErr {
|
||||
Logger.debug("Successfully retrieved device format", context: ["attempt": String(attempt)])
|
||||
return streamFormat
|
||||
}
|
||||
|
||||
Logger.info(
|
||||
"------- Failed to get stream format after device ready check, retrying...",
|
||||
context: [
|
||||
"attempt": String(attempt),
|
||||
"max_retries": String(maxRetries),
|
||||
"status": String(status),
|
||||
"device_id": String(deviceID),
|
||||
])
|
||||
|
||||
// Don't delay on the last attempt
|
||||
if attempt < maxRetries {
|
||||
Thread.sleep(forTimeInterval: Double(retryDelayMs) / 1000.0)
|
||||
}
|
||||
}
|
||||
|
||||
// If all attempts failed after device readiness confirmation, this is a genuine error
|
||||
Logger.error(
|
||||
"Failed to get device format after device readiness check and retries",
|
||||
context: [
|
||||
"device_id": String(deviceID),
|
||||
"device_was_ready": "true",
|
||||
])
|
||||
|
||||
fatalError(
|
||||
"Failed to get stream format from ready device: \(deviceID). This indicates a Core Audio subsystem error."
|
||||
)
|
||||
}
|
||||
|
||||
static func createMetadata(for format: AudioStreamBasicDescription) -> AudioStreamMetadata {
|
||||
return AudioStreamMetadata(
|
||||
sampleRate: format.mSampleRate,
|
||||
channelsPerFrame: format.mChannelsPerFrame,
|
||||
bitsPerChannel: format.mBitsPerChannel,
|
||||
isFloat: format.mFormatFlags & kAudioFormatFlagIsFloat != 0,
|
||||
captureMode: "audio",
|
||||
deviceName: nil, // TODO: Get device name if needed
|
||||
deviceUID: nil, // TODO: Get device UID if needed
|
||||
encoding: format.mFormatFlags & kAudioFormatFlagIsFloat != 0 ? "pcm_f32le" : "pcm_s16le"
|
||||
)
|
||||
}
|
||||
|
||||
public static func writeMetadata(for format: AudioStreamBasicDescription) {
|
||||
let metadata = createMetadata(for: format)
|
||||
Logger.writeMessage(.metadata, data: metadata)
|
||||
Logger.writeMessage(.streamStart, data: Optional<String>.none)
|
||||
}
|
||||
|
||||
public static func logFormatInfo(_ format: AudioStreamBasicDescription) {
|
||||
Logger.debug(
|
||||
"Using device's native format",
|
||||
context: [
|
||||
"channels": String(format.mChannelsPerFrame),
|
||||
"sample_rate": String(format.mSampleRate),
|
||||
"bits_per_channel": String(format.mBitsPerChannel),
|
||||
"format_id": String(format.mFormatID),
|
||||
"format_flags": String(format: "0x%08x", format.mFormatFlags),
|
||||
"bytes_per_frame": String(format.mBytesPerFrame),
|
||||
]
|
||||
)
|
||||
}
|
||||
}
|
||||
@@ -3,18 +3,15 @@ import Foundation
|
||||
public struct AudioPacket {
|
||||
public let timestamp: Date
|
||||
public let duration: Double
|
||||
public let peakAmplitude: Float // useful for level monitoring
|
||||
public let rawAudioData: Data
|
||||
public let data: Data
|
||||
|
||||
public init(
|
||||
timestamp: Date,
|
||||
duration: Double,
|
||||
peakAmplitude: Float,
|
||||
rawAudioData: Data
|
||||
data: Data
|
||||
) {
|
||||
self.timestamp = timestamp
|
||||
self.duration = duration
|
||||
self.peakAmplitude = peakAmplitude
|
||||
self.rawAudioData = rawAudioData
|
||||
self.data = data
|
||||
}
|
||||
}
|
||||
@@ -5,24 +5,23 @@ import Foundation
|
||||
public class AudioRecorder {
|
||||
private var deviceID: AudioObjectID
|
||||
private var ioProcID: AudioDeviceIOProcID?
|
||||
private var sourceFormat: AudioStreamBasicDescription?
|
||||
private var finalFormat: AudioStreamBasicDescription?
|
||||
private var finalFormat: AudioStreamBasicDescription!
|
||||
private var audioBuffer: AudioBuffer?
|
||||
private var outputHandler: AudioOutputHandler
|
||||
private var converter: AudioFormatConverter?
|
||||
private var chunkDuration: Double
|
||||
|
||||
init(
|
||||
public init(
|
||||
deviceID: AudioObjectID, outputHandler: AudioOutputHandler, convertToSampleRate: Double? = nil,
|
||||
chunkDuration: Double = 0.2
|
||||
) {
|
||||
self.deviceID = deviceID
|
||||
self.outputHandler = outputHandler
|
||||
self.chunkDuration = chunkDuration
|
||||
|
||||
// Get source format and set up conversion if requested
|
||||
let sourceFormat = AudioFormatManager.getDeviceFormat(deviceID: deviceID)
|
||||
self.sourceFormat = sourceFormat
|
||||
|
||||
// Set up the audio buffer using source format and configurable chunk duration
|
||||
self.audioBuffer = AudioBuffer(format: sourceFormat, chunkDuration: chunkDuration)
|
||||
|
||||
if let targetSampleRate = convertToSampleRate {
|
||||
// Validate sample rate
|
||||
@@ -52,29 +51,22 @@ public class AudioRecorder {
|
||||
}
|
||||
}
|
||||
|
||||
func startRecording() {
|
||||
public func startRecording() {
|
||||
Logger.debug("Starting audio recording")
|
||||
|
||||
guard let sourceFormat = sourceFormat, let finalFormat = finalFormat else {
|
||||
fatalError("Audio formats not initialized")
|
||||
}
|
||||
|
||||
// Set up the audio buffer using source format and configurable chunk duration
|
||||
self.audioBuffer = AudioBuffer(format: sourceFormat, chunkDuration: chunkDuration)
|
||||
|
||||
// Log format info and send metadata for FINAL format
|
||||
// Log format info and send metadata for final format
|
||||
AudioFormatManager.logFormatInfo(finalFormat)
|
||||
let metadata = AudioFormatManager.createMetadata(for: finalFormat)
|
||||
outputHandler.handleMetadata(metadata)
|
||||
outputHandler.handleStreamStart()
|
||||
|
||||
// Set up and start the IO proc
|
||||
setupAndStartIOProc()
|
||||
|
||||
Logger.info("Audio device started successfully")
|
||||
}
|
||||
|
||||
// FIXME: note to self, what about installTap? Would require audio engine and a node?
|
||||
// Note to self, what about installTap? Would require audio engine and a node?
|
||||
// No; AudioEngine.installTap() can only fire as often as 100ms. too slow for us
|
||||
private func setupAndStartIOProc() {
|
||||
Logger.debug("Creating IO proc")
|
||||
var status = AudioDeviceCreateIOProcID(
|
||||
@@ -115,24 +107,23 @@ public class AudioRecorder {
|
||||
let audioData = Data(bytes: firstBuffer.mData!, count: Int(firstBuffer.mDataByteSize))
|
||||
audioBuffer?.append(audioData)
|
||||
|
||||
processAudioBuffer()
|
||||
|
||||
return noErr
|
||||
}
|
||||
|
||||
public func stopRecording() {
|
||||
processAudioBuffer()
|
||||
outputHandler.handleStreamStop()
|
||||
cleanupIOProc()
|
||||
}
|
||||
|
||||
private func processAudioBuffer() {
|
||||
// Process and send complete chunks, applying conversion if needed
|
||||
audioBuffer?.processChunks().forEach { packet in
|
||||
let processedPacket = converter?.transform(packet) ?? packet
|
||||
outputHandler.handleAudioPacket(processedPacket)
|
||||
}
|
||||
|
||||
return noErr
|
||||
}
|
||||
|
||||
func stopRecording() {
|
||||
// Send any remaining buffered audio, applying conversion if needed
|
||||
if let finalPacket = audioBuffer?.flushRemaining() {
|
||||
let processedPacket = converter?.transform(finalPacket) ?? finalPacket
|
||||
outputHandler.handleAudioPacket(processedPacket)
|
||||
}
|
||||
|
||||
outputHandler.handleStreamStop()
|
||||
cleanupIOProc()
|
||||
}
|
||||
|
||||
private func cleanupIOProc() {
|
||||
+10
-16
@@ -3,14 +3,12 @@ import AudioToolbox
|
||||
import CoreAudio
|
||||
import Foundation
|
||||
|
||||
class AudioTapManager {
|
||||
public class AudioTapManager {
|
||||
private var tapID: AudioObjectID?
|
||||
private var deviceID: AudioObjectID?
|
||||
|
||||
init() {
|
||||
// Empty init - setup happens in setupAudioTap()
|
||||
}
|
||||
|
||||
public init() {}
|
||||
|
||||
deinit {
|
||||
Logger.debug("Cleaning up audio tap manager")
|
||||
|
||||
@@ -26,7 +24,7 @@ class AudioTapManager {
|
||||
}
|
||||
|
||||
/// Sets up the audio tap and aggregate device
|
||||
func setupAudioTap(with config: TapConfiguration) throws {
|
||||
public func setupAudioTap(with config: TapConfiguration) throws {
|
||||
Logger.debug("Setting up audio tap manager")
|
||||
|
||||
tapID = try createSystemAudioTap(with: config)
|
||||
@@ -42,34 +40,30 @@ class AudioTapManager {
|
||||
}
|
||||
|
||||
/// Returns the aggregate device ID for recording
|
||||
func getDeviceID() -> AudioObjectID? {
|
||||
public func getDeviceID() -> AudioObjectID? {
|
||||
return deviceID
|
||||
}
|
||||
|
||||
private func createSystemAudioTap(with config: TapConfiguration) throws -> AudioObjectID {
|
||||
Logger.debug("Creating tap description")
|
||||
// Create a tap description
|
||||
let description = CATapDescription()
|
||||
|
||||
// Configure the tap to capture all system audio
|
||||
description.name = "audiotee-tap"
|
||||
description.processes = try translatePIDsToProcessObjects(config.processes) // Properly translate PIDs
|
||||
description.isPrivate = true
|
||||
description.muteBehavior = config.muteBehavior.coreAudioValue
|
||||
description.isMixdown = true
|
||||
description.isMono = true
|
||||
description.isMixdown = true
|
||||
description.isMono = config.isMono
|
||||
description.isExclusive = config.isExclusive
|
||||
description.deviceUID = nil // system default
|
||||
description.stream = 0 // first stream of output device
|
||||
description.deviceUID = nil // system default
|
||||
description.stream = 0 // first stream of output device
|
||||
|
||||
Logger.debug(
|
||||
"Tap description configured",
|
||||
context: [
|
||||
"name": description.name,
|
||||
"processes": String(describing: config.processes),
|
||||
"private": String(description.isPrivate),
|
||||
"mute": String(describing: description.muteBehavior),
|
||||
"mixdown": String(description.isMixdown),
|
||||
"mono": String(description.isMono),
|
||||
"exclusive": String(description.isExclusive),
|
||||
])
|
||||
@@ -153,4 +147,4 @@ class AudioTapManager {
|
||||
throw AudioTeeError.tapAssignmentFailed(status)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+3
-1
@@ -2,10 +2,12 @@ public struct TapConfiguration {
|
||||
public let processes: [Int32]
|
||||
public let muteBehavior: TapMuteBehavior
|
||||
public let isExclusive: Bool
|
||||
public let isMono: Bool
|
||||
|
||||
public init(processes: [Int32], muteBehavior: TapMuteBehavior, isExclusive: Bool) {
|
||||
public init(processes: [Int32], muteBehavior: TapMuteBehavior, isExclusive: Bool, isMono: Bool) {
|
||||
self.processes = processes
|
||||
self.muteBehavior = muteBehavior
|
||||
self.isExclusive = isExclusive
|
||||
self.isMono = isMono
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,6 @@
|
||||
import ArgumentParser
|
||||
import CoreAudio
|
||||
|
||||
public enum TapMuteBehavior: String, CaseIterable, ExpressibleByArgument {
|
||||
public enum TapMuteBehavior: String, CaseIterable {
|
||||
case unmuted = "unmuted"
|
||||
case muted = "muted"
|
||||
|
||||
-1
@@ -1,4 +1,3 @@
|
||||
|
||||
import Foundation
|
||||
|
||||
/// Protocol for handling audio output in different formats
|
||||
+8
-9
@@ -1,18 +1,17 @@
|
||||
import Foundation
|
||||
|
||||
/// Binary output with JSON headers (pipe-optimised)
|
||||
public class BinaryAudioOutputHandler: AudioOutputHandler {
|
||||
public init() {}
|
||||
private let flushAfterWrite: Bool
|
||||
|
||||
public init(flushAfterWrite: Bool = false) {
|
||||
self.flushAfterWrite = flushAfterWrite
|
||||
}
|
||||
public func handleAudioPacket(_ packet: AudioPacket) {
|
||||
// Create metadata without the audio data
|
||||
let metadata = BinaryPacketHeader(from: packet)
|
||||
|
||||
// Write JSON metadata line
|
||||
Logger.writeMessage(.audio, data: metadata)
|
||||
|
||||
// Write raw binary audio data directly to stdout
|
||||
FileHandle.standardOutput.write(packet.rawAudioData)
|
||||
FileHandle.standardOutput.write(packet.data)
|
||||
if flushAfterWrite {
|
||||
fflush(stdout)
|
||||
}
|
||||
}
|
||||
|
||||
public func handleMetadata(_ metadata: AudioStreamMetadata) {
|
||||
@@ -7,9 +7,6 @@ public enum MessageType: String, Codable {
|
||||
case streamStart = "stream_start"
|
||||
case streamStop = "stream_stop"
|
||||
|
||||
// Audio data
|
||||
case audio
|
||||
|
||||
// Logging
|
||||
case info
|
||||
case error
|
||||
@@ -24,8 +24,8 @@ public class Logger {
|
||||
let message = Message(type: type, data: data)
|
||||
do {
|
||||
let jsonData = try jsonEncoder.encode(message)
|
||||
FileHandle.standardOutput.write(jsonData)
|
||||
FileHandle.standardOutput.write("\n".data(using: .utf8)!)
|
||||
FileHandle.standardError.write(jsonData)
|
||||
FileHandle.standardError.write("\n".data(using: .utf8)!)
|
||||
} catch {
|
||||
// TODO: handle at some point
|
||||
}
|
||||
@@ -89,9 +89,9 @@ func translatePIDsToProcessObjects(_ pids: [Int32]) throws -> [AudioObjectID] {
|
||||
}
|
||||
|
||||
extension String {
|
||||
func print(to fileHandle: FileHandle) {
|
||||
if let data = (self + "\n").data(using: .utf8) {
|
||||
fileHandle.write(data)
|
||||
}
|
||||
func print(to fileHandle: FileHandle) {
|
||||
if let data = (self + "\n").data(using: .utf8) {
|
||||
fileHandle.write(data)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,161 +0,0 @@
|
||||
import ArgumentParser
|
||||
import CoreAudio
|
||||
import Foundation
|
||||
|
||||
struct AudioTee: ParsableCommand {
|
||||
static let configuration = CommandConfiguration(
|
||||
abstract: "Capture system audio and stream to stdout",
|
||||
discussion: """
|
||||
AudioTee captures system audio using Core Audio taps and streams it as structured output.
|
||||
|
||||
Output formats:
|
||||
• json: Base64-encoded audio in JSON messages (safe for terminals)
|
||||
• binary: Raw binary audio with JSON metadata headers (efficient for pipes)
|
||||
• auto: Automatically choose based on whether stdout is a terminal (default)
|
||||
|
||||
Process filtering:
|
||||
• include-processes: Only tap specified process IDs (empty = all processes)
|
||||
• exclude-processes: Tap all processes except specified ones
|
||||
• mute: How to handle processes being tapped
|
||||
|
||||
Examples:
|
||||
audiotee # Auto format, tap all processes
|
||||
audiotee --format=json # Always use JSON format
|
||||
audiotee --format=binary # Always use binary format
|
||||
audiotee --sample-rate=16000 # Convert to 16kHz mono for ASR
|
||||
audiotee --sample-rate=8000 # Convert to 8kHz for telephony
|
||||
audiotee --include-processes 1234 # Only tap process 1234
|
||||
audiotee --include-processes 1234 5678 9012 # Tap only these processes
|
||||
audiotee --exclude-processes 1234 5678 # Tap everything except these
|
||||
audiotee --mute # Mute processes being tapped
|
||||
"""
|
||||
)
|
||||
|
||||
@Option(name: .shortAndLong, help: "Output format")
|
||||
var format: OutputFormat = .auto
|
||||
|
||||
@Option(
|
||||
name: .long, help: "Process IDs to include (space-separated, empty = all processes)")
|
||||
var includeProcesses: [Int32] = []
|
||||
|
||||
@Option(
|
||||
name: .long, help: "Process IDs to exclude (space-separated)")
|
||||
var excludeProcesses: [Int32] = []
|
||||
|
||||
@Flag(name: .long, help: "Mute processes being tapped")
|
||||
var mute: Bool = false
|
||||
|
||||
@Option(
|
||||
name: .long,
|
||||
help: "Target sample rate (8000, 16000, 22050, 24000, 32000, 44100, 48000)")
|
||||
var sampleRate: Double?
|
||||
|
||||
@Option(
|
||||
name: .long,
|
||||
help: "Audio chunk duration in seconds (default: 0.2)")
|
||||
var chunkDuration: Double = 0.2
|
||||
|
||||
func validate() throws {
|
||||
if !includeProcesses.isEmpty && !excludeProcesses.isEmpty {
|
||||
throw ValidationError("Cannot specify both --include-processes and --exclude-processes")
|
||||
}
|
||||
}
|
||||
|
||||
func run() throws {
|
||||
setupSignalHandlers()
|
||||
|
||||
Logger.info("Starting AudioTee...")
|
||||
Logger.debug("Using output format: \(format)")
|
||||
|
||||
// Validate chunk duration
|
||||
guard chunkDuration > 0 && chunkDuration <= 5.0 else {
|
||||
Logger.error(
|
||||
"Invalid chunk duration",
|
||||
context: ["chunk_duration": String(chunkDuration), "valid_range": "0.0 < duration <= 5.0"])
|
||||
throw ExitCode.failure
|
||||
}
|
||||
|
||||
// Convert include/exclude processes to TapConfiguration format
|
||||
let (processes, isExclusive) = convertProcessFlags()
|
||||
|
||||
let tapConfig = TapConfiguration(
|
||||
processes: processes,
|
||||
muteBehavior: mute ? .muted : .unmuted,
|
||||
isExclusive: isExclusive
|
||||
)
|
||||
|
||||
let audioTapManager = AudioTapManager()
|
||||
do {
|
||||
try audioTapManager.setupAudioTap(with: tapConfig)
|
||||
} catch AudioTeeError.pidTranslationFailed(let failedPIDs) {
|
||||
Logger.error(
|
||||
"Failed to translate process IDs to audio objects",
|
||||
context: [
|
||||
"failed_pids": failedPIDs.map(String.init).joined(separator: ", "),
|
||||
"suggestion": "Check that the process IDs exist and are running",
|
||||
])
|
||||
throw ExitCode.failure
|
||||
} catch {
|
||||
Logger.error(
|
||||
"Failed to setup audio tap", context: ["error": String(describing: error)])
|
||||
throw ExitCode.failure
|
||||
}
|
||||
|
||||
guard let deviceID = audioTapManager.getDeviceID() else {
|
||||
Logger.error("Failed to get device ID from audio tap manager")
|
||||
throw ExitCode.failure
|
||||
}
|
||||
|
||||
let outputHandler = createOutputHandler(for: format)
|
||||
let recorder = AudioRecorder(
|
||||
deviceID: deviceID, outputHandler: outputHandler, convertToSampleRate: sampleRate,
|
||||
chunkDuration: chunkDuration)
|
||||
recorder.startRecording()
|
||||
|
||||
// Run until the run loop is stopped (by signal handler)
|
||||
while true {
|
||||
let result = CFRunLoopRunInMode(CFRunLoopMode.defaultMode, 0.1, false)
|
||||
if result == CFRunLoopRunResult.stopped || result == CFRunLoopRunResult.finished {
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
Logger.info("Shutting down...")
|
||||
recorder.stopRecording()
|
||||
}
|
||||
|
||||
private func setupSignalHandlers() {
|
||||
signal(SIGINT) { _ in
|
||||
Logger.info("Received SIGINT, initiating graceful shutdown...")
|
||||
CFRunLoopStop(CFRunLoopGetMain())
|
||||
}
|
||||
signal(SIGTERM) { _ in
|
||||
Logger.info("Received SIGTERM, initiating graceful shutdown...")
|
||||
CFRunLoopStop(CFRunLoopGetMain())
|
||||
}
|
||||
}
|
||||
|
||||
private func createOutputHandler(for format: OutputFormat) -> AudioOutputHandler {
|
||||
switch format {
|
||||
case .json:
|
||||
return JSONAudioOutputHandler()
|
||||
case .binary:
|
||||
return BinaryAudioOutputHandler()
|
||||
case .auto:
|
||||
return AutoAudioOutputHandler()
|
||||
}
|
||||
}
|
||||
|
||||
private func convertProcessFlags() -> ([Int32], Bool) {
|
||||
if !includeProcesses.isEmpty {
|
||||
// Include specific processes only
|
||||
return (includeProcesses, false)
|
||||
} else if !excludeProcesses.isEmpty {
|
||||
// Exclude specific processes (tap everything except these)
|
||||
return (excludeProcesses, true)
|
||||
} else {
|
||||
// Default: tap everything
|
||||
return ([], true)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,18 +0,0 @@
|
||||
import ArgumentParser
|
||||
|
||||
enum OutputFormat: String, CaseIterable, ExpressibleByArgument {
|
||||
case json = "json"
|
||||
case binary = "binary"
|
||||
case auto = "auto"
|
||||
|
||||
var description: String {
|
||||
switch self {
|
||||
case .json:
|
||||
return "Base64-encoded JSON (terminal-safe)"
|
||||
case .binary:
|
||||
return "Binary with JSON headers (pipe-optimised)"
|
||||
case .auto:
|
||||
return "Auto-detect based on TTY (default)"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,61 +0,0 @@
|
||||
import CoreAudio
|
||||
import Foundation
|
||||
|
||||
public class AudioBuffer {
|
||||
private var buffer = Data()
|
||||
private let targetChunkDuration: Double
|
||||
private let streamFormat: AudioStreamBasicDescription
|
||||
|
||||
public init(format: AudioStreamBasicDescription, chunkDuration: Double = 0.2) {
|
||||
self.streamFormat = format
|
||||
self.targetChunkDuration = chunkDuration
|
||||
}
|
||||
|
||||
public func append(_ data: Data) {
|
||||
buffer.append(data)
|
||||
}
|
||||
|
||||
public func processChunks() -> [AudioPacket] {
|
||||
var packets: [AudioPacket] = []
|
||||
|
||||
while let packet = nextChunk() {
|
||||
packets.append(packet)
|
||||
}
|
||||
|
||||
return packets
|
||||
}
|
||||
|
||||
public func flushRemaining() -> AudioPacket? {
|
||||
guard !buffer.isEmpty else { return nil }
|
||||
|
||||
let packet = AudioPacket(
|
||||
timestamp: Date(),
|
||||
duration: 0.0, // Unknown duration for final chunk
|
||||
peakAmplitude: 0.0,
|
||||
rawAudioData: buffer
|
||||
)
|
||||
|
||||
buffer.removeAll()
|
||||
return packet
|
||||
}
|
||||
|
||||
private func nextChunk() -> AudioPacket? {
|
||||
let bytesPerFrame = Int(streamFormat.mBytesPerFrame)
|
||||
let samplesPerChunk = Int(streamFormat.mSampleRate * targetChunkDuration)
|
||||
let bytesPerChunk = samplesPerChunk * bytesPerFrame
|
||||
|
||||
guard buffer.count >= bytesPerChunk else { return nil }
|
||||
|
||||
let chunkData = buffer.prefix(bytesPerChunk)
|
||||
|
||||
let packet = AudioPacket(
|
||||
timestamp: Date(),
|
||||
duration: Double(samplesPerChunk) / streamFormat.mSampleRate,
|
||||
peakAmplitude: 0.0, // No analysis in raw mode
|
||||
rawAudioData: Data(chunkData)
|
||||
)
|
||||
|
||||
buffer.removeFirst(bytesPerChunk)
|
||||
return packet
|
||||
}
|
||||
}
|
||||
@@ -1,54 +0,0 @@
|
||||
import AudioToolbox
|
||||
import CoreAudio
|
||||
import Foundation
|
||||
|
||||
public class AudioFormatManager {
|
||||
public static func getDeviceFormat(deviceID: AudioObjectID) -> AudioStreamBasicDescription {
|
||||
var propertyAddress = getPropertyAddress(
|
||||
selector: kAudioDevicePropertyStreamFormat,
|
||||
scope: kAudioDevicePropertyScopeInput)
|
||||
var propertySize = UInt32(MemoryLayout<AudioStreamBasicDescription>.stride)
|
||||
var streamFormat = AudioStreamBasicDescription()
|
||||
let status = AudioObjectGetPropertyData(
|
||||
deviceID, &propertyAddress, 0, nil, &propertySize, &streamFormat)
|
||||
|
||||
guard status == noErr else {
|
||||
fatalError("Failed to get stream format: \(status)")
|
||||
}
|
||||
|
||||
return streamFormat
|
||||
}
|
||||
|
||||
static func createMetadata(for format: AudioStreamBasicDescription) -> AudioStreamMetadata {
|
||||
return AudioStreamMetadata(
|
||||
sampleRate: format.mSampleRate,
|
||||
channelsPerFrame: format.mChannelsPerFrame,
|
||||
bitsPerChannel: format.mBitsPerChannel,
|
||||
isFloat: format.mFormatFlags & kAudioFormatFlagIsFloat != 0,
|
||||
captureMode: "audio",
|
||||
deviceName: nil, // TODO: Get device name if needed
|
||||
deviceUID: nil, // TODO: Get device UID if needed
|
||||
encoding: format.mFormatFlags & kAudioFormatFlagIsFloat != 0 ? "pcm_f32le" : "pcm_s16le"
|
||||
)
|
||||
}
|
||||
|
||||
public static func writeMetadata(for format: AudioStreamBasicDescription) {
|
||||
let metadata = createMetadata(for: format)
|
||||
Logger.writeMessage(.metadata, data: metadata)
|
||||
Logger.writeMessage(.streamStart, data: Optional<String>.none)
|
||||
}
|
||||
|
||||
public static func logFormatInfo(_ format: AudioStreamBasicDescription) {
|
||||
Logger.debug(
|
||||
"Using device's native format",
|
||||
context: [
|
||||
"channels": String(format.mChannelsPerFrame),
|
||||
"sample_rate": String(format.mSampleRate),
|
||||
"bits_per_channel": String(format.mBitsPerChannel),
|
||||
"format_id": String(format.mFormatID),
|
||||
"format_flags": String(format: "0x%08x", format.mFormatFlags),
|
||||
"bytes_per_frame": String(format.mBytesPerFrame),
|
||||
]
|
||||
)
|
||||
}
|
||||
}
|
||||
@@ -1,31 +0,0 @@
|
||||
import Foundation
|
||||
|
||||
/// Auto-detecting output handler based on TTY
|
||||
public class AutoAudioOutputHandler: AudioOutputHandler {
|
||||
private let handler: AudioOutputHandler
|
||||
|
||||
public init() {
|
||||
// Auto-detect based on whether stdout is a terminal
|
||||
if isatty(STDOUT_FILENO) != 0 {
|
||||
handler = JSONAudioOutputHandler()
|
||||
} else {
|
||||
handler = BinaryAudioOutputHandler()
|
||||
}
|
||||
}
|
||||
|
||||
public func handleAudioPacket(_ packet: AudioPacket) {
|
||||
handler.handleAudioPacket(packet)
|
||||
}
|
||||
|
||||
public func handleMetadata(_ metadata: AudioStreamMetadata) {
|
||||
handler.handleMetadata(metadata)
|
||||
}
|
||||
|
||||
public func handleStreamStart() {
|
||||
handler.handleStreamStart()
|
||||
}
|
||||
|
||||
public func handleStreamStop() {
|
||||
handler.handleStreamStop()
|
||||
}
|
||||
}
|
||||
@@ -1,23 +0,0 @@
|
||||
import Foundation
|
||||
|
||||
/// Base64-encoded JSON output (terminal-safe)
|
||||
public class JSONAudioOutputHandler: AudioOutputHandler {
|
||||
public init() {}
|
||||
|
||||
public func handleAudioPacket(_ packet: AudioPacket) {
|
||||
let jsonPacket = JSONAudioPacket(from: packet)
|
||||
Logger.writeMessage(.audio, data: jsonPacket)
|
||||
}
|
||||
|
||||
public func handleMetadata(_ metadata: AudioStreamMetadata) {
|
||||
Logger.writeMessage(.metadata, data: metadata)
|
||||
}
|
||||
|
||||
public func handleStreamStart() {
|
||||
Logger.writeMessage(.streamStart, data: Optional<String>.none)
|
||||
}
|
||||
|
||||
public func handleStreamStop() {
|
||||
Logger.writeMessage(.streamStop, data: Optional<String>.none)
|
||||
}
|
||||
}
|
||||
@@ -1,45 +0,0 @@
|
||||
import Foundation
|
||||
|
||||
/// JSON-serializable version of AudioPacket with base64-encoded audio data
|
||||
public struct JSONAudioPacket: Codable {
|
||||
public let timestamp: Date
|
||||
public let duration: Double
|
||||
public let peakAmplitude: Float
|
||||
public let audioData: String // base64 encoded audio data
|
||||
|
||||
public enum CodingKeys: String, CodingKey {
|
||||
case timestamp
|
||||
case duration
|
||||
case peakAmplitude = "peak_amplitude"
|
||||
case audioData = "audio_data"
|
||||
}
|
||||
|
||||
public init(from packet: AudioPacket) {
|
||||
self.timestamp = packet.timestamp
|
||||
self.duration = packet.duration
|
||||
self.peakAmplitude = packet.peakAmplitude
|
||||
self.audioData = packet.rawAudioData.base64EncodedString()
|
||||
}
|
||||
}
|
||||
|
||||
/// Metadata-only packet for binary output (without base64 audio data)
|
||||
public struct BinaryPacketHeader: Codable {
|
||||
public let timestamp: Date
|
||||
public let duration: Double
|
||||
public let peakAmplitude: Float
|
||||
public let audioLength: Int // Length of raw audio data in bytes
|
||||
|
||||
public enum CodingKeys: String, CodingKey {
|
||||
case timestamp
|
||||
case duration
|
||||
case peakAmplitude = "peak_amplitude"
|
||||
case audioLength = "audio_length"
|
||||
}
|
||||
|
||||
public init(from packet: AudioPacket) {
|
||||
self.timestamp = packet.timestamp
|
||||
self.duration = packet.duration
|
||||
self.peakAmplitude = packet.peakAmplitude
|
||||
self.audioLength = packet.rawAudioData.count
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
import XCTest
|
||||
@testable import AudioTeeCore
|
||||
|
||||
final class AudioPacketTests: XCTestCase {
|
||||
func testPacketCreation() {
|
||||
let timestamp = Date()
|
||||
let duration = 1.0
|
||||
let data = Data([0x01, 0x02, 0x03, 0x04])
|
||||
|
||||
let packet = AudioPacket(
|
||||
timestamp: timestamp,
|
||||
duration: duration,
|
||||
data: data
|
||||
)
|
||||
|
||||
XCTAssertEqual(packet.timestamp, timestamp)
|
||||
XCTAssertEqual(packet.duration, duration)
|
||||
XCTAssertEqual(packet.data, data)
|
||||
}
|
||||
|
||||
func testPacketDataSize() {
|
||||
let packet = AudioPacket(
|
||||
timestamp: Date(),
|
||||
duration: 0.5,
|
||||
data: Data(repeating: 0xFF, count: 1024)
|
||||
)
|
||||
|
||||
XCTAssertEqual(packet.data.count, 1024)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user