refactor: rename to common/, clients/desktop, clients/wearos
Move three directories into their target topology: - doctate-common/ -> common/ - client-desktop/ -> clients/desktop/ - watch/wearos/ -> clients/wearos/ Update doctate-common path-dependency in 3 Cargo.toml files (server, clients/desktop, experiments) and the workspace members list in the root Cargo.toml. Crate names unchanged (doctate-server, doctate-common, doctate-desktop). Workspace still in single-Cargo.lock form; isolation into standalone workspaces follows in stage 3. All 508 tests pass, experiments standalone-workspace also resolves the new path.
This commit is contained in:
@@ -0,0 +1,288 @@
|
||||
//! Recording-side domain types shared between server and clients.
|
||||
//!
|
||||
//! Each recording has a single sidecar JSON next to its audio file:
|
||||
//! `<stem>.json` (suffix [`RECORDING_META_SUFFIX`]). It carries the
|
||||
//! container duration and the transcript outcome in one structured
|
||||
//! document.
|
||||
//!
|
||||
//! On-disk vs. in-memory representation:
|
||||
//!
|
||||
//! - [`Transcript`] is the **on-disk** shape. It only models the two
|
||||
//! *terminal* outcomes the transcriber can persist (`Silent` or
|
||||
//! `Content`). A recording that has not been transcribed yet has no
|
||||
//! sidecar at all — there is no "pending" variant on disk.
|
||||
//! - [`TranscriptState`] is the **in-memory** shape returned by the
|
||||
//! server's reader. It adds a [`TranscriptState::Pending`] variant
|
||||
//! for "no sidecar exists", so callers can `match` on the three
|
||||
//! logically distinct states without having to inspect the
|
||||
//! filesystem themselves.
|
||||
//!
|
||||
//! Call-sites that collapse this onto a 2-state view (`Option<String>`,
|
||||
//! `.exists()`, `is_empty()`) drift apart and have caused real bugs:
|
||||
//! e.g. the UI treating `Silent` like `Content` while the oneliner
|
||||
//! worker treated it like `Pending`, leaving cases stuck in
|
||||
//! "Generating" forever.
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
/// Filename suffix of the per-recording metadata JSON sidecar.
|
||||
///
|
||||
/// Used as `<audio_stem>.<RECORDING_META_SUFFIX>` next to the `.m4a`,
|
||||
/// e.g. `2026-04-19T10-00-00Z.json`. Centralised here so the server
|
||||
/// reader, worker writer, and recovery scan agree on one location.
|
||||
pub const RECORDING_META_SUFFIX: &str = "json";
|
||||
|
||||
/// On-disk transcript outcome.
|
||||
///
|
||||
/// Internally tagged: serializes as `{"state": "silent"}` or
|
||||
/// `{"state": "content", "text": "..."}` — the discriminator lives
|
||||
/// inside the JSON object next to the data, so the wire format stays
|
||||
/// flat. Only the two terminal variants the transcriber can produce
|
||||
/// appear here; "not transcribed yet" is represented by the absence
|
||||
/// of the sidecar file (see [`TranscriptState::Pending`]).
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
#[serde(tag = "state", rename_all = "lowercase")]
|
||||
pub enum Transcript {
|
||||
/// Transcriber ran and classified the audio as silence — terminal.
|
||||
/// The next transcribe pass will not re-run unless the user
|
||||
/// re-uploads.
|
||||
Silent,
|
||||
/// Transcriber produced non-empty text.
|
||||
Content { text: String },
|
||||
}
|
||||
|
||||
/// Replay-gain data for one recording, derived from a single
|
||||
/// `volumedetect` pass at the end of the transcribe pipeline.
|
||||
///
|
||||
/// `gain_db` is applied browser-side via a Web Audio `GainNode` so
|
||||
/// recordings from clients with very different mic levels (Watch
|
||||
/// ~-38 dB mean, desktop ~-27 dB) play back at a consistent loudness;
|
||||
/// the audio file itself is never modified. `mean_db` and `max_db` are
|
||||
/// persisted alongside so the gain target can be retuned later without
|
||||
/// re-decoding the audio.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
|
||||
pub struct Loudness {
|
||||
pub mean_db: f64,
|
||||
pub max_db: f64,
|
||||
pub gain_db: f64,
|
||||
}
|
||||
|
||||
/// Per-recording metadata persisted as `<stem>.json`.
|
||||
///
|
||||
/// Single atomic write at the end of the transcribe pipeline. The
|
||||
/// presence of this file is the canonical signal "the transcriber
|
||||
/// finished for this recording". Absent file = still pending.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
|
||||
pub struct RecordingMeta {
|
||||
/// Transcriber outcome. See [`Transcript`].
|
||||
pub transcript: Transcript,
|
||||
/// Container duration in whole seconds, from `ffprobe` on the
|
||||
/// remuxed audio. `None` is allowed so test fixtures and recovery
|
||||
/// flows that mint a metadata file without invoking ffprobe stay
|
||||
/// valid; production worker writes always populate this.
|
||||
pub duration_seconds: Option<u32>,
|
||||
/// Replay-gain data. `None` for sidecars written before this field
|
||||
/// existed and for production paths where `volumedetect` failed.
|
||||
/// `#[serde(default)]` keeps older JSON without this field parsable.
|
||||
#[serde(default)]
|
||||
pub loudness: Option<Loudness>,
|
||||
}
|
||||
|
||||
/// Three-way state of a recording's transcript, as seen by the server
|
||||
/// after consulting the on-disk sidecar.
|
||||
///
|
||||
/// Introduced to force every call-site to decide explicitly how it
|
||||
/// treats `Silent` recordings via `match` exhaustiveness. Unlike
|
||||
/// [`Transcript`], this carries a [`Self::Pending`] variant — the
|
||||
/// reader synthesises that when the sidecar JSON does not exist.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum TranscriptState {
|
||||
/// No sidecar JSON yet — transcriber hasn't produced a result.
|
||||
/// Callers should typically wait or enqueue.
|
||||
Pending,
|
||||
/// JSON exists with `state: "silent"`. Terminal: the transcriber
|
||||
/// already ran and classified this recording as silence.
|
||||
Silent,
|
||||
/// JSON exists with `state: "content"`.
|
||||
Content { text: String },
|
||||
}
|
||||
|
||||
impl From<Transcript> for TranscriptState {
|
||||
fn from(t: Transcript) -> Self {
|
||||
match t {
|
||||
Transcript::Silent => Self::Silent,
|
||||
Transcript::Content { text } => Self::Content { text },
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl TranscriptState {
|
||||
/// Build a `TranscriptState` from a parsed [`RecordingMeta`] (or its
|
||||
/// absence). `None` (file missing or malformed) becomes
|
||||
/// [`Self::Pending`].
|
||||
pub fn from_meta(meta: Option<&RecordingMeta>) -> Self {
|
||||
match meta {
|
||||
None => Self::Pending,
|
||||
Some(m) => m.transcript.clone().into(),
|
||||
}
|
||||
}
|
||||
|
||||
/// True when the transcriber has finished for this recording,
|
||||
/// regardless of whether it produced content or silence. Use this
|
||||
/// for UI "is the transcription done?" checks — `Silent` counts.
|
||||
pub fn has_file(&self) -> bool {
|
||||
!matches!(self, Self::Pending)
|
||||
}
|
||||
|
||||
/// True when there is no sidecar yet and the transcriber still
|
||||
/// owes us a result. Use this for "should I enqueue transcription?"
|
||||
/// or "is a oneliner worth deferring?" gates.
|
||||
pub fn is_pending(&self) -> bool {
|
||||
matches!(self, Self::Pending)
|
||||
}
|
||||
|
||||
/// Non-empty transcript text, if any. Returns `None` for both
|
||||
/// `Pending` and `Silent` — callers that need to distinguish must
|
||||
/// `match` directly.
|
||||
pub fn as_content(&self) -> Option<&str> {
|
||||
match self {
|
||||
Self::Content { text } => Some(text.as_str()),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn missing_meta_is_pending() {
|
||||
assert_eq!(TranscriptState::from_meta(None), TranscriptState::Pending);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn silent_meta_maps_to_silent_state() {
|
||||
let meta = RecordingMeta {
|
||||
transcript: Transcript::Silent,
|
||||
duration_seconds: Some(7),
|
||||
loudness: None,
|
||||
};
|
||||
assert_eq!(
|
||||
TranscriptState::from_meta(Some(&meta)),
|
||||
TranscriptState::Silent
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn content_meta_maps_to_content_state() {
|
||||
let meta = RecordingMeta {
|
||||
transcript: Transcript::Content {
|
||||
text: "Hallo".into(),
|
||||
},
|
||||
duration_seconds: None,
|
||||
loudness: None,
|
||||
};
|
||||
assert_eq!(
|
||||
TranscriptState::from_meta(Some(&meta)),
|
||||
TranscriptState::Content {
|
||||
text: "Hallo".into()
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn has_file_distinguishes_pending() {
|
||||
assert!(!TranscriptState::Pending.has_file());
|
||||
assert!(TranscriptState::Silent.has_file());
|
||||
assert!(TranscriptState::Content { text: "x".into() }.has_file());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn as_content_only_for_content() {
|
||||
assert_eq!(TranscriptState::Pending.as_content(), None);
|
||||
assert_eq!(TranscriptState::Silent.as_content(), None);
|
||||
assert_eq!(
|
||||
TranscriptState::Content { text: "x".into() }.as_content(),
|
||||
Some("x")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn transcript_silent_serializes_with_tag() {
|
||||
let json = serde_json::to_string(&Transcript::Silent).unwrap();
|
||||
assert_eq!(json, r#"{"state":"silent"}"#);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn transcript_content_serializes_with_tag_and_text() {
|
||||
let json = serde_json::to_string(&Transcript::Content {
|
||||
text: "Hallo".into(),
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(json, r#"{"state":"content","text":"Hallo"}"#);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recording_meta_roundtrip_with_duration() {
|
||||
let meta = RecordingMeta {
|
||||
transcript: Transcript::Content {
|
||||
text: "Befund.".into(),
|
||||
},
|
||||
duration_seconds: Some(42),
|
||||
loudness: None,
|
||||
};
|
||||
let json = serde_json::to_string(&meta).unwrap();
|
||||
let back: RecordingMeta = serde_json::from_str(&json).unwrap();
|
||||
assert_eq!(back, meta);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recording_meta_roundtrip_without_duration() {
|
||||
let meta = RecordingMeta {
|
||||
transcript: Transcript::Silent,
|
||||
duration_seconds: None,
|
||||
loudness: None,
|
||||
};
|
||||
let json = serde_json::to_string(&meta).unwrap();
|
||||
assert!(json.contains(r#""duration_seconds":null"#));
|
||||
let back: RecordingMeta = serde_json::from_str(&json).unwrap();
|
||||
assert_eq!(back, meta);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recording_meta_roundtrip_with_loudness() {
|
||||
let meta = RecordingMeta {
|
||||
transcript: Transcript::Content {
|
||||
text: "Befund.".into(),
|
||||
},
|
||||
duration_seconds: Some(42),
|
||||
loudness: Some(Loudness {
|
||||
mean_db: -27.3,
|
||||
max_db: -3.1,
|
||||
gain_db: 2.1,
|
||||
}),
|
||||
};
|
||||
let json = serde_json::to_string(&meta).unwrap();
|
||||
let back: RecordingMeta = serde_json::from_str(&json).unwrap();
|
||||
assert_eq!(back, meta);
|
||||
}
|
||||
|
||||
/// Backwards-compat: a sidecar written before `loudness` existed
|
||||
/// must still parse, with the field defaulting to `None`. This is
|
||||
/// what `#[serde(default)]` on the field guarantees and is the only
|
||||
/// reason we don't need a migration step for pre-loudness recordings.
|
||||
#[test]
|
||||
fn old_meta_without_loudness_parses_as_none() {
|
||||
let json = r#"{"transcript":{"state":"silent"},"duration_seconds":7}"#;
|
||||
let meta: RecordingMeta = serde_json::from_str(json).unwrap();
|
||||
assert_eq!(
|
||||
meta,
|
||||
RecordingMeta {
|
||||
transcript: Transcript::Silent,
|
||||
duration_seconds: Some(7),
|
||||
loudness: None,
|
||||
}
|
||||
);
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user