Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
81 changes: 81 additions & 0 deletions crates/aster-cli/src/dictate.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,81 @@
//! `aster dictate`: record until stdin gets a line or closes, then print the
//! transcript. Front-ends with no chat turn running use it for their mic button.

use std::io::BufRead;

use anyhow::Result;
use aster_voice::{Recording, Transcriber, VoiceError};
use serde_json::json;

/// What went wrong, in one plain sentence, with the raw detail kept apart.
#[derive(Clone, Debug)]
pub(crate) struct DictationFailure {
pub(crate) message: &'static str,
pub(crate) detail: Option<String>,
}

impl DictationFailure {
pub(crate) fn no_key() -> Self {
Self {
message: "Voice input needs a speech key. Run `aster key set ELEVENLABS_API_KEY` (or OPENAI_API_KEY), then try again.",
detail: None,
}
}

pub(crate) fn interrupted(detail: String) -> Self {
Self {
message: "Recording stopped unexpectedly. Try again.",
detail: Some(detail),
}
}
}

impl From<VoiceError> for DictationFailure {
fn from(err: VoiceError) -> Self {
let (message, detail) = match err {
VoiceError::Unsupported => ("Voice input isn't available on this system yet.", None),
VoiceError::NoMicrophone(detail) => (
"Can't reach your microphone. Check that Aster has microphone access in System Settings, then try again.",
Some(detail),
),
VoiceError::TooShort => (
"That was too short to hear. Speak for a moment before stopping.",
None,
),
VoiceError::Provider { provider, detail } => (
"Couldn't turn your recording into text. Check your speech key and connection, then try again.",
Some(format!("{provider}: {detail}")),
),
};
Self { message, detail }
}
}

pub(crate) async fn run() -> Result<()> {
let result = dictate().await;
let event = match result {
Ok(text) => json!({ "type": "transcript", "text": text }),
Err(failure) => json!({
"type": "error",
"message": failure.message,
"detail": failure.detail,
}),
};
println!("{event}");
Ok(())
}

async fn dictate() -> Result<String, DictationFailure> {
let transcriber = Transcriber::from_env().ok_or_else(DictationFailure::no_key)?;
let recording = Recording::start()?;
println!("{}", json!({ "type": "listening" }));
tokio::task::spawn_blocking(|| std::io::stdin().lock().read_line(&mut String::new()))
.await
.map_err(|e| DictationFailure::interrupted(e.to_string()))?
.map_err(|e| DictationFailure::interrupted(e.to_string()))?;
println!("{}", json!({ "type": "transcribing" }));
let clip = tokio::task::spawn_blocking(|| recording.finish())
.await
.map_err(|e| DictationFailure::interrupted(e.to_string()))?;
Ok(transcriber.transcribe(&clip).await?)
}
4 changes: 4 additions & 0 deletions crates/aster-cli/src/main.rs
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@ mod cloudflare_auth;
mod config;
mod credentials;
mod cron;
mod dictate;
mod edits;
mod fix;
mod git;
Expand Down Expand Up @@ -164,6 +165,8 @@ enum Command {
Announce(announce::AnnounceArgs),
/// Set a one-shot native notification: `aster remind "text" "in 10s"`.
Remind(remind::RemindArgs),
/// Record from the microphone until stdin gets a line, then print the transcript as NDJSON.
Dictate,
/// Score the last turn of a session and refine the learned skill for that task.
Learn(learn::LearnArgs),
/// Run Python 3 with the standard library built in: `aster python script.py` or `-c "..."`.
Expand Down Expand Up @@ -258,6 +261,7 @@ async fn main() -> Result<()> {
Command::Cron(args) => cron::run(args),
Command::Announce(args) => announce::run(args).await,
Command::Remind(args) => remind::run(args),
Command::Dictate => dictate::run().await,
Command::Learn(args) => learn::run(args).await,
#[cfg(target_os = "android")]
Command::Python(args) => python::run(args),
Expand Down
3 changes: 2 additions & 1 deletion crates/aster-cli/src/tui/chat.rs
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,7 @@ use super::bottom_pane::{
BottomPane, CommandDesc, InputResult, ModelPickerView, SelectionItem, UnifiedItem,
UnifiedSection, scan_mentions,
};
use super::dictation::{Dictation, DictationFailure};
use super::dictation::Dictation;
use super::guard::TuiGuard;
use super::helpers::{clip_row, count_of, human_count, listed, short_path};
use super::markdown::{self, MarkdownStream};
Expand All @@ -33,6 +33,7 @@ use super::{history, theme, wrap};
use crate::chat::{
Answer, ApprovalRequest, QuestionRequest, Resume, SessionCtx, UiRequest, UiSender,
};
use crate::dictate::DictationFailure;
use crate::persist::Recorder;

type ChatTurn = tokio::task::JoinHandle<Result<(String, Vec<String>, Option<Vec<ChatMessage>>)>>;
Expand Down
48 changes: 9 additions & 39 deletions crates/aster-cli/src/tui/dictation.rs
Original file line number Diff line number Diff line change
@@ -1,10 +1,11 @@
//! Ctrl+R dictation in the chat composer. Terminals do not report key release,
//! so one press starts listening and the next sends the clip to be transcribed.

use aster_voice::{Recording, Transcriber, VoiceError};
use aster_voice::{Recording, Transcriber};
use tokio::sync::mpsc;

use super::chat::AppEvent;
use crate::dictate::DictationFailure;

#[derive(Default)]
pub(super) enum Dictation {
Expand All @@ -14,39 +15,27 @@ pub(super) enum Dictation {
Transcribing,
}

/// What went wrong, in one plain sentence, with the raw detail kept apart.
#[derive(Clone)]
pub(super) struct DictationFailure {
pub(super) message: &'static str,
pub(super) detail: Option<String>,
}

impl Dictation {
pub(super) fn toggle(
&mut self,
tx: &mpsc::UnboundedSender<AppEvent>,
) -> Result<(), DictationFailure> {
match std::mem::take(self) {
Self::Idle => {
let Some(transcriber) = Transcriber::from_env() else {
return Err(DictationFailure {
message: "Voice input needs a speech key. Run `aster key set ELEVENLABS_API_KEY` (or OPENAI_API_KEY), then try again.",
detail: None,
});
};
let recording = Recording::start().map_err(failure)?;
let transcriber = Transcriber::from_env().ok_or_else(DictationFailure::no_key)?;
let recording = Recording::start()?;
*self = Self::Listening(recording, transcriber);
}
Self::Listening(recording, transcriber) => {
*self = Self::Transcribing;
let tx = tx.clone();
tokio::spawn(async move {
let result = match tokio::task::spawn_blocking(|| recording.finish()).await {
Ok(clip) => transcriber.transcribe(&clip).await.map_err(failure),
Err(err) => Err(DictationFailure {
message: "Recording stopped unexpectedly. Try again.",
detail: Some(err.to_string()),
}),
Ok(clip) => transcriber
.transcribe(&clip)
.await
.map_err(DictationFailure::from),
Err(err) => Err(DictationFailure::interrupted(err.to_string())),
};
let _ = tx.send(AppEvent::Dictated(result));
});
Expand All @@ -64,22 +53,3 @@ impl Dictation {
}
}
}

fn failure(err: VoiceError) -> DictationFailure {
let (message, detail) = match err {
VoiceError::Unsupported => ("Voice input isn't available on this system yet.", None),
VoiceError::NoMicrophone(detail) => (
"Can't reach your microphone. Check that your terminal has microphone access in System Settings, then try again.",
Some(detail),
),
VoiceError::TooShort => (
"That was too short to hear. Speak for a moment before pressing ctrl+r again.",
None,
),
VoiceError::Provider { provider, detail } => (
"Couldn't turn your recording into text. Check your speech key and connection, then try again.",
Some(format!("{provider}: {detail}")),
),
};
DictationFailure { message, detail }
}
8 changes: 8 additions & 0 deletions desktop/src-tauri/Info.plist
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
<key>NSMicrophoneUsageDescription</key>
<string>Aster uses the microphone when you turn on dictation, so you can speak a message instead of typing it.</string>
</dict>
</plist>
2 changes: 2 additions & 0 deletions desktop/src-tauri/entitlements.plist
Original file line number Diff line number Diff line change
Expand Up @@ -12,5 +12,7 @@
<true/>
<key>com.apple.security.network.client</key>
<true/>
<key>com.apple.security.device.audio-input</key>
<true/>
</dict>
</plist>
75 changes: 75 additions & 0 deletions desktop/src-tauri/src/dictation.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,75 @@
//! The composer's mic button runs one `aster dictate` child at a time. Its
//! NDJSON goes to the UI as `aster://dictation`; stopping writes it a newline.

use std::process::Stdio;
use std::sync::Mutex;

use tauri::{AppHandle, Emitter};
use tokio::io::{AsyncBufReadExt, AsyncWriteExt, BufReader};
use tokio::process::{Child, ChildStdin, Command};

struct Active {
child: Child,
stdin: Option<ChildStdin>,
}

static ACTIVE: Mutex<Option<Active>> = Mutex::new(None);

#[tauri::command]
pub(crate) async fn start_dictation(app: AppHandle) -> Result<(), String> {
let mut cmd = Command::new(crate::resolve_bin());
crate::hide_console(&mut cmd);
cmd.arg("dictate")
.env("PATH", crate::augmented_path())
.stdin(Stdio::piped())
.stdout(Stdio::piped())
.stderr(Stdio::null());
let mut child = cmd
.spawn()
.map_err(|e| format!("could not launch aster: {e}"))?;
let stdin = child.stdin.take();
let stdout = child.stdout.take().ok_or("aster dictate has no stdout")?;
let previous = ACTIVE
.lock()
.map_err(|e| e.to_string())?
.replace(Active { child, stdin });
if let Some(mut previous) = previous {
let _ = previous.child.start_kill();
}

tauri::async_runtime::spawn(async move {
let mut lines = BufReader::new(stdout).lines();
while let Ok(Some(line)) = lines.next_line().await {
if !line.trim().is_empty() {
let _ = app.emit("aster://dictation", line);
}
}
let _ = app.emit("aster://dictation", r#"{"type":"closed"}"#);
});
Ok(())
}

#[tauri::command]
pub(crate) async fn stop_dictation() -> Result<(), String> {
let stdin = ACTIVE
.lock()
.map_err(|e| e.to_string())?
.as_mut()
.and_then(|active| active.stdin.take());
let Some(mut stdin) = stdin else {
return Ok(());
};
stdin
.write_all(b"\n")
.await
.map_err(|e| format!("could not stop the recording: {e}"))
}

#[tauri::command]
pub(crate) async fn cancel_dictation() -> Result<(), String> {
let active = ACTIVE.lock().map_err(|e| e.to_string())?.take();
if let Some(mut active) = active {
let _ = active.child.start_kill();
}
Ok(())
}
5 changes: 5 additions & 0 deletions desktop/src-tauri/src/lib.rs
Original file line number Diff line number Diff line change
@@ -1,3 +1,5 @@
mod dictation;

use std::path::PathBuf;
use std::process::Stdio;
use std::sync::{Arc, Mutex};
Expand Down Expand Up @@ -971,6 +973,9 @@ pub fn run() {
chat,
answer_approval,
cancel_chat,
dictation::start_dictation,
dictation::stop_dictation,
dictation::cancel_dictation,
apply_fix,
list_sessions,
show_session,
Expand Down
33 changes: 33 additions & 0 deletions desktop/src/components/Composer.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@ import { useEffect, useMemo, useRef, useState, type ReactElement, type ReactNode
import type { Effort, PermissionMode, Provider, ReviewOpts, SourceKind } from "../lib/types";
import { SOURCE_LABELS } from "../lib/session";
import { listRepoFiles } from "../lib/aster";
import { useDictation } from "../lib/dictation";
import { applyTrigger, dropTrigger, triggersAt, type Trigger } from "../lib/trigger";
import { useListNav } from "../lib/listnav";
import { EFFORT_OPTIONS, effortShort } from "../lib/effort";
Expand All @@ -10,6 +11,7 @@ import { ApprovalPicker, permissionIcon, permissionLabel } from "./ApprovalPicke
import { Autocomplete, type Suggestion } from "./Autocomplete";
import { ChoiceList } from "./ChoiceList";
import { CommandMenu, type MenuItem, type MenuSection } from "./CommandMenu";
import { useToast } from "./chrome";
import { ModelMenu } from "./ModelMenu";
import { Popover } from "./Popover";
import { IconMorphGlyph, sendStop } from "../interior/icon-morph";
Expand All @@ -26,6 +28,7 @@ import {
GearIcon,
GitCommitIcon,
GitPullRequestIcon,
MicIcon,
NewChatIcon,
PlusIcon,
ReviewIcon,
Expand All @@ -36,6 +39,12 @@ import {

const MAX_ROWS = 10;

const micTitle = {
idle: "Dictate a message",
listening: "Stop and turn into text (Esc discards)",
transcribing: "Turning your recording into text…",
} as const;

type Menu = "none" | "add" | "commands" | "permission" | "settings" | "model" | "provider" | "project" | "source";

const ICONS: Record<string, ReactElement> = {
Expand Down Expand Up @@ -236,6 +245,13 @@ export function Composer({
focusInput(at);
};

const toast = useToast();
const dictation = useDictation((heard) => {
const before = text.slice(0, caret);
const gap = before && !/\s$/.test(before) ? " " : "";
write(before + gap + heard + text.slice(caret), caret + gap.length + heard.length);
}, toast);

const send = () => {
const trimmed = text.trim();
if (busy) return;
Expand Down Expand Up @@ -386,6 +402,10 @@ export function Composer({
}, [menuOpen, model, providers, effort, permissionMode, intent, repoName, onCommand, onEffort, onOpenSettings]);

const onKeyDown = (e: React.KeyboardEvent<HTMLTextAreaElement>) => {
if (e.key === "Escape" && dictation.state === "listening") {
dictation.cancel();
return;
}
if (command) return;
if (suggestions.length > 0 && files.onKey(e)) return;
if (e.key === "Enter" && !e.shiftKey) {
Expand Down Expand Up @@ -628,6 +648,19 @@ export function Composer({
<span>{intent === "review" ? "Review" : permissionLabel(permissionMode)}</span>
</button>

{intent === "chat" && (
<button
className={dictation.state === "idle" ? "ghost foot-btn" : "ghost foot-btn mic-on"}
onClick={dictation.toggle}
disabled={dictation.state === "transcribing"}
title={micTitle[dictation.state]}
aria-label={micTitle[dictation.state]}
aria-pressed={dictation.state === "listening"}
>
<MicIcon />
</button>
)}

<button
className={busy ? "send stop" : "send"}
onClick={busy ? onCancel : send}
Expand Down
Loading
Loading