diff --git a/bazel/python/deps.bzl b/bazel/python/deps.bzl index f8ee6c24..1ff2f03a 100644 --- a/bazel/python/deps.bzl +++ b/bazel/python/deps.bzl @@ -14,13 +14,15 @@ entirely in `pyproject.toml`. This is a list of *which marker- conditional names rules_python skips*, derived from grepping requirements.lock.txt for `; python_full_version` / `; sys_platform` markers that don't satisfy our pinned 3.12 + cross-platform set. +Korean G2P's two platform-specific analyzer roots also need an explicit +select because the hub omits them from all_requirements on matching hosts. Track upstream: bazel-contrib/rules_python#2244 (and friends) — once fixed, this whole file collapses to `_RUNTIME_DEPS = all_requirements` directly in the consumer BUILD. """ -load("@pypi//:requirements.bzl", _all_requirements = "all_requirements") +load("@pypi//:requirements.bzl", _all_requirements = "all_requirements", _requirement = "requirement") # Names that appear in `all_requirements` but whose BUILD file pip.parse # elides because of a `python_full_version` / `sys_platform` marker. @@ -64,6 +66,9 @@ _MARKER_FILTERED = [ # sys_platform == 'win32' "pywin32_ctypes", "tzdata", + # Korean G2P's platform-specific roots are selected explicitly below. + "eunjeon", + "python_mecab_ko", ] def _is_filtered(label): @@ -75,8 +80,13 @@ def _is_filtered(label): def all_runtime_deps(): """Every dep the lockfile resolves for the current platform. - No name list maintained anywhere in BUILD/justfile/MODULE — call this - from py_library/py_binary/py_test `deps =` and the dependency set is - implicit in pyproject.toml + requirements.lock.txt. + Call this from py_library/py_binary/py_test `deps =`. Packages come from + pyproject.toml + requirements.lock.txt, with the Korean analyzer selected + for the target platform below. """ - return [d for d in _all_requirements if not _is_filtered(d)] + # The hub omits these marker-conditional roots from all_requirements even + # on matching hosts. Keep them available to g2pk2 without runtime pip calls. + return [d for d in _all_requirements if not _is_filtered(d)] + select({ + "@platforms//os:windows": [_requirement("eunjeon")], + "//conditions:default": [_requirement("python-mecab-ko")], + }) diff --git a/book/src/batchalign/user-guide/cli-reference.md b/book/src/batchalign/user-guide/cli-reference.md index 519a4fd2..aa598601 100644 --- a/book/src/batchalign/user-guide/cli-reference.md +++ b/book/src/batchalign/user-guide/cli-reference.md @@ -50,6 +50,7 @@ or over their input as described below. See [Command I/O](../reference/command-i |---|---| | `transcribe` | Recording to CHAT transcript | | `align` | Forced alignment of CHAT against audio | +| `phonetic` | Add observed IPA to `%pho` using audio and phone-sequence DP | | `morphotag` | Add `%mor` and `%gra` | | `utseg` | Revise utterance segmentation | | `translate` | Add translation tiers | @@ -143,6 +144,83 @@ Accepts the shared input selection and `-o/--out` options above. | `--engine` | pyannote-ai | Choices: `pyannote-ai`, `pyannote`. Diarization engine: pyannote-ai (cloud) or pyannote (local). | | `--num-speakers`, `-n` | 0 | Expected speaker count; zero auto-detects. | +## phonetic + +Accepts timed CHAT and matching audio, using shared input selection and +`-o/--out`. Install the `phonetic` extra for PhoneticXeus, Piper Plus G2P, and Epitran. +Phonetic transcription requires Python 3.11 or newer. +The first inference downloads the pinned model revision; +building the CLI and displaying help do not download model weights. + +```bash +just batchalign cli phonetic recording.cha --out phonetic-output --force-cpu +``` + +The command recognizes phones from audio and DP-aligns them against reference +IPA pronunciations to recover word boundaries. `%pho` retains the observed IPA, +including pronunciation differences. Existing `%pho` tiers are preserved. +The input must have utterance timing bullets; use `utr` first when needed. +Word-level forced alignment is not required. + +Like Whisper forced alignment, phonetic inference groups consecutive utterances +into approximately 20-second audio windows, then projects phones back to their +original words and utterances. Gaps over two seconds and backwards timings start +a new window; an utterance longer than 20 seconds stays whole. Words receiving +no phones are marked `…` (uncoded) and reported with their utterance timing. + +Windows of similar duration are batched using padding and real encoder lengths. +Normalization is performed independently per window, and padded output frames +are excluded from decoding. CPU defaults to one window at a time; CUDA defaults +to two. Increase `--batch-size` to try higher GPU throughput, or decrease it to +reduce memory use. Batched and single-window output can differ slightly because +the upstream encoder's convolution branches remain sensitive to padding. + +| Option | Default | Details | +|---|---|---| +| `--pronunciations` | None | UTF-8 CSV with header `word,ipa`; one word or whole phonological unit and its IPA per row, overriding the generated pronunciation. | +| `--force-cpu` | False | Use CPU instead of automatic CUDA selection. MPS is not selected. | +| `--batch-size` | CPU 1, CUDA 2 | Maximum audio windows per model batch; must be positive. | + +The task runner passes the primary `@Languages` code from CHAT. No separate +language option is needed. [Piper Plus G2P](https://pypi.org/project/piper-plus-g2p/) +0.2.0 handles English, Japanese, Mandarin Chinese, Korean, Spanish, French, +Portuguese, and Swedish. Both `cmn` and `zho` select Mandarin; Cantonese (`yue`) +is a separate language and uses the fallback. + +Other languages use [Epitran](https://github.com/dmort27/epitran), with their +default script resolved automatically (e.g. Russian `rus-Cyrl`, Hindi +`hin-Deva`). A failure in a supported Piper backend is reported rather than +silently switching providers. English does not require Flite's `lex_lookup` +or eSpeak; Python language packages are included in the extra. Language +resources may download on first use. + +The DP compares IPA directly, preserving distinctions such as nasalization, +vowel length, and tone. Piper's language-specific phone/tone labels are +converted to IPA before comparison. References only determine grouping; +they never replace the observed phones. Alternate scripts and code-switched +words can use pronunciation overrides. +Pass `--pronunciations pronunciations.csv` to supply dialect forms or words +in unsupported languages. For example: + +```csv +word,ipa +wug,wʌɡ +bonjour,bɔ̃ʒuʁ +the cat,ðəkæt +``` + +The header is required. Use standard CSV quoting for cells containing commas. +Words are matched without case; empty cells and duplicate words are errors. +An unknown pronunciation, invalid +audio window, or alignment leaving a word without phones fails the file without +overwriting it. Insertions between word anchors attach to the preceding word; +review inferred boundaries, especially around reduced or atypical speech. + +Python pipelines can compose `recipes.phonetic(phonetic_backend=backend, +utr_backend=...)` with existing tasks. Custom backends implement the `Phonetic` +marker and typed `PhoneticInput`/`PhoneticOutput` contract; the Rust runner owns +CHAT extraction, result validation, and tier insertion. + ## ai Accepts the shared input selection and `-o/--out` options above. diff --git a/crates/batchalign/batchalign-core/src/base.rs b/crates/batchalign/batchalign-core/src/base.rs index 7901f21e..f4f7f3c1 100644 --- a/crates/batchalign/batchalign-core/src/base.rs +++ b/crates/batchalign/batchalign-core/src/base.rs @@ -13,6 +13,7 @@ use crate::proto::convert::{ConvertInput, MediaOutput}; use crate::proto::coref::{CorefInput, CorefOutput}; use crate::proto::fa::{FaInput, FaOutput}; use crate::proto::morphosyntax::{MorphosyntaxInput, MorphosyntaxOutput}; +use crate::proto::phonetic::{PhoneticInput, PhoneticOutput}; use crate::proto::speaker::{SpeakerInput, SpeakerOutput}; use crate::proto::translate::{TranslateInput, TranslateOutput}; use crate::proto::utr::{UtrInput, UtrOutput}; @@ -69,6 +70,8 @@ pub enum Task { Compare, /// Decode media and encode a new WAV or MP3 artifact. Convert, + /// Acoustic phonetic transcription into `%pho`. + Phonetic, } impl Task { @@ -91,6 +94,7 @@ impl Task { // already get bullets from UtSeg.) Task::Fa => &[Task::UtSeg, Task::Utr], Task::Morphosyntax => &[Task::UtSeg], + Task::Phonetic => &[Task::UtSeg, Task::Utr, Task::Fa], Task::Coref => &[Task::Morphosyntax], Task::Translate => &[Task::Morphosyntax], Task::Compare => &[], @@ -109,6 +113,7 @@ impl Task { Task::Utr => "utr", Task::Morphosyntax => "morphosyntax", Task::Translate => "translate", + Task::Phonetic => "phonetic", Task::Coref => "coref", Task::Compare => "compare", Task::Convert => "convert", @@ -116,7 +121,7 @@ impl Task { } /// Every variant — useful for iteration in tests and codegen. - pub const ALL: [Task; 11] = [ + pub const ALL: [Task; 12] = [ Task::Ai, Task::Asr, Task::Fa, @@ -125,6 +130,7 @@ impl Task { Task::Utr, Task::Morphosyntax, Task::Translate, + Task::Phonetic, Task::Coref, Task::Compare, Task::Convert, @@ -200,6 +206,7 @@ union_input_output! { Ai(AiInput) => Ai, Asr(AsrInput) => Asr, Fa(FaInput) => Fa, + Phonetic(PhoneticInput) => Phonetic, Speaker(SpeakerInput) => Speaker, UtSeg(UtSegInput) => UtSeg, // UTR's payload is serde-transparent over `AsrInput`, so the @@ -217,6 +224,7 @@ union_input_output! { Ai(AiOutput), Asr(AsrOutput), Fa(FaOutput), + Phonetic(PhoneticOutput), Speaker(SpeakerOutput), UtSeg(UtSegOutput), Utr(UtrOutput), @@ -252,6 +260,7 @@ try_from_output! { Ai(AiOutput), Asr(AsrOutput), Fa(FaOutput), + Phonetic(PhoneticOutput), Speaker(SpeakerOutput), UtSeg(UtSegOutput), Utr(UtrOutput), diff --git a/crates/batchalign/batchalign-core/src/lib.rs b/crates/batchalign/batchalign-core/src/lib.rs index 05aef663..a7fae820 100644 --- a/crates/batchalign/batchalign-core/src/lib.rs +++ b/crates/batchalign/batchalign-core/src/lib.rs @@ -39,6 +39,9 @@ pub use base::{ pub use cache::CacheKey; pub use metrics::{MetricsArtifact, MetricsKind, MetricsRow, MetricsTable}; pub use proto::convert::{ConvertInput, MediaFormat, MediaOutput}; +pub use proto::phonetic::{ + PhoneticInput, PhoneticOutput, PhoneticResult, PhoneticUnit, PhoneticUtterance, +}; pub use utils::{ AiChatInput, AudioError, BAError, BAResult, ChatInput, MediaInput, PairedInput, PreparedAudio, SourceId, SpeakerLabel, prepare_pcm, prepare_pcm_interleaved, diff --git a/crates/batchalign/batchalign-core/src/proto/mod.rs b/crates/batchalign/batchalign-core/src/proto/mod.rs index 35e2ce01..40efb6e5 100644 --- a/crates/batchalign/batchalign-core/src/proto/mod.rs +++ b/crates/batchalign/batchalign-core/src/proto/mod.rs @@ -28,6 +28,7 @@ pub mod compare; pub mod convert; pub mod coref; pub mod fa; +pub mod phonetic; pub mod morphosyntax; pub mod speaker; pub mod translate; diff --git a/crates/batchalign/batchalign-core/src/proto/phonetic.rs b/crates/batchalign/batchalign-core/src/proto/phonetic.rs new file mode 100644 index 00000000..b97f0ee4 --- /dev/null +++ b/crates/batchalign/batchalign-core/src/proto/phonetic.rs @@ -0,0 +1,101 @@ +//! Acoustic phonetic transcription. CHAT unit ownership survives inference. + +use crate::cache::{CacheKey, hash_serialized}; +use crate::utils::{PreparedAudio, SourceId}; +use schemars::JsonSchema; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +pub struct PhoneticUnit { + /// Original spoken text, or a CHAT pause to preserve structurally. + pub text: String, + pub pause: bool, +} + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +pub struct PhoneticUtterance { + /// Line index in the source AST; results must echo it in order. + pub index: usize, + pub start_ms: u64, + pub end_ms: u64, + pub units: Vec, +} + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +pub struct PhoneticInput { + pub source_id: SourceId, + pub audio: PreparedAudio, + pub language: String, + pub utterances: Vec, +} + +impl CacheKey for PhoneticInput { + fn hash(&self, hasher: &mut blake3::Hasher) { + // Crop bounds and unit ownership affect both inference and projection. + hash_serialized(&(&self.audio, &self.language, &self.utterances), hasher); + } +} + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +pub struct PhoneticResult { + pub index: usize, + /// One IPA string per input unit; pauses must be echoed unchanged. + pub ipa: Vec, +} + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +pub struct PhoneticOutput { + pub source_id: SourceId, + pub utterances: Vec, +} + +crate::register_proto_schema!(PhoneticUnit); +crate::register_proto_schema!(PhoneticUtterance); +crate::register_proto_schema!(PhoneticInput); +crate::register_proto_schema!(PhoneticResult); +crate::register_proto_schema!(PhoneticOutput); + +#[cfg(test)] +mod tests { + use super::*; + + fn digest(input: &PhoneticInput) -> blake3::Hash { + let mut hasher = blake3::Hasher::new(); + input.hash(&mut hasher); + hasher.finalize() + } + + #[test] + fn cache_tracks_audio_windows_and_reference_but_not_source_path() { + let mut input = PhoneticInput { + source_id: SourceId::try_new("a.cha").unwrap(), + audio: PreparedAudio { + pcm_f32le: vec![0; 64], + sample_rate: 16000, + channels: 1, + frame_count: 16, + }, + language: "eng".into(), + utterances: vec![PhoneticUtterance { + index: 0, + start_ms: 0, + end_ms: 1, + units: vec![PhoneticUnit { + text: "cat".into(), + pause: false, + }], + }], + }; + let original = digest(&input); + input.source_id = SourceId::try_new("b.cha").unwrap(); + assert_eq!(digest(&input), original); + input.utterances[0].end_ms = 2; + assert_ne!(digest(&input), original); + input.utterances[0].end_ms = 1; + input.utterances[0].units[0].text = "dog".into(); + assert_ne!(digest(&input), original); + input.utterances[0].units[0].text = "cat".into(); + input.audio.pcm_f32le[0] = 1; + assert_ne!(digest(&input), original); + } +} diff --git a/crates/batchalign/batchalign-core/src/taskrunners/mod.rs b/crates/batchalign/batchalign-core/src/taskrunners/mod.rs index 29aec4dd..d637ec23 100644 --- a/crates/batchalign/batchalign-core/src/taskrunners/mod.rs +++ b/crates/batchalign/batchalign-core/src/taskrunners/mod.rs @@ -10,6 +10,7 @@ pub mod convert; pub mod coref; pub mod fa; mod media; +pub mod phonetic; pub mod morphosyntax; pub mod speaker; pub mod translate; @@ -31,6 +32,7 @@ pub fn canonical(task: Task) -> Box { Task::Ai => Box::new(ai::AiTaskRunner::default()), Task::Asr => Box::new(asr::AsrTaskRunner), Task::Fa => Box::new(fa::FaTaskRunner), + Task::Phonetic => Box::new(phonetic::PhoneticTaskRunner), Task::Speaker => Box::new(speaker::SpeakerTaskRunner), Task::UtSeg => Box::new(utseg::UtSegTaskRunner), Task::Morphosyntax => Box::new(morphosyntax::MorphosyntaxTaskRunner), diff --git a/crates/batchalign/batchalign-core/src/taskrunners/phonetic.rs b/crates/batchalign/batchalign-core/src/taskrunners/phonetic.rs new file mode 100644 index 00000000..d96f7319 --- /dev/null +++ b/crates/batchalign/batchalign-core/src/taskrunners/phonetic.rs @@ -0,0 +1,254 @@ +//! Extract phonological units, dispatch acoustic inference, and add typed `%pho`. + +use crate::base::{ + BAValue, Chat, Dispatcher, ProgressEvent, ProgressSink, ScaledProgress, Task, TaskInput, + TaskRunner, +}; +use crate::proto::phonetic::{PhoneticInput, PhoneticOutput, PhoneticUnit, PhoneticUtterance}; +use crate::utils::{BAError, BAResult, prepare_pcm}; +use async_trait::async_trait; +use std::sync::Arc; +use talkbank_model::alignment::helpers::{TierDomain, collect_tier_items}; +use talkbank_model::{DependentTier, Line, PhoTier, PhoTierType}; + +pub struct PhoneticTaskRunner; + +#[async_trait] +impl TaskRunner for PhoneticTaskRunner { + const TASK: Task = Task::Phonetic; + + async fn apply( + &self, + value: &mut BAValue, + dispatcher: &dyn Dispatcher, + sink: Arc, + ) -> BAResult<()> { + let chat = match value { + BAValue::Chat(chat) => chat, + BAValue::Failed { .. } => return Ok(()), + other => { + return Err(BAError::Validation(format!( + "phonetic requires CHAT, got {}", + other.kind() + ))); + } + }; + let utterances = extract_utterances(chat)?; + if utterances.is_empty() { + return Ok(()); + } + let media = chat + .media() + .cloned() + .or_else(|| super::media::sibling_media(chat)) + .ok_or_else(|| { + BAError::Validation("phonetic requires @Media or sibling audio".into()) + })?; + let input = PhoneticInput { + source_id: chat.source_id().clone(), + audio: prepare_pcm(&media) + .map_err(|err| BAError::Internal(format!("audio_prep: {err:#}")))?, + language: chat.primary_language().unwrap_or_default(), + utterances: utterances.clone(), + }; + sink.emit(ProgressEvent::stage_started( + chat.source_id(), + Task::Phonetic, + )); + let progress = Arc::new(ScaledProgress::new( + sink.clone(), + chat.source_id().clone(), + Task::Phonetic, + 1, + )); + progress.start_step(); + let output: PhoneticOutput = dispatcher + .dispatch_with_progress(TaskInput::Phonetic(input), progress) + .await? + .try_into()?; + inject_tiers(chat, &utterances, output)?; + sink.emit(ProgressEvent::stage_injected( + chat.source_id(), + Task::Phonetic, + )); + Ok(()) + } +} + +fn extract_utterances(chat: &Chat) -> BAResult> { + let mut utterances = Vec::new(); + for (index, line) in chat.ast().lines.as_slice().iter().enumerate() { + let Line::Utterance(utterance) = line else { + continue; + }; + if utterance + .dependent_tiers + .iter() + .any(|entry| matches!(&entry.tier, DependentTier::Pho(_))) + { + continue; + } + let units: Vec<_> = collect_tier_items(&utterance.main.content.content, TierDomain::Pho) + .into_iter() + .map(|position| PhoneticUnit { + pause: position.description.as_deref() == Some("pause"), + text: position.text, + }) + .collect(); + if units.is_empty() { + continue; + } + let timing = utterance + .main + .content + .bullet + .as_ref() + .map(|bullet| &bullet.timing) + .filter(|timing| timing.end_ms > timing.start_ms) + .ok_or_else(|| { + BAError::Validation(format!( + "phonetic: line {} needs an utterance timing bullet; run utr first", + index + 1 + )) + })?; + utterances.push(PhoneticUtterance { + index, + start_ms: timing.start_ms, + end_ms: timing.end_ms, + units, + }); + } + Ok(utterances) +} + +fn inject_tiers( + chat: &mut Chat, + inputs: &[PhoneticUtterance], + output: PhoneticOutput, +) -> BAResult<()> { + if inputs.len() != output.utterances.len() { + return Err(BAError::Worker( + "phonetic: utterance/output count mismatch".into(), + )); + } + // Validate every result before changing the original AST. + let mut revised = Chat::from_validated_ast(chat.ast().clone(), chat.source_id().clone()); + if let Some(media) = chat.media().cloned() { + revised = revised.with_media(media); + } + for (input, result) in inputs.iter().zip(output.utterances) { + if result.index != input.index || result.ipa.len() != input.units.len() { + return Err(BAError::Worker( + "phonetic: result unit ownership mismatch".into(), + )); + } + for (unit, ipa) in input.units.iter().zip(&result.ipa) { + if (unit.pause && ipa != &unit.text) + || ipa.is_empty() + || ipa.chars().any(|ch| ch.is_whitespace() || ch.is_control()) + { + return Err(BAError::Worker(format!( + "phonetic: invalid IPA unit {ipa:?}" + ))); + } + } + let Line::Utterance(utterance) = &mut revised.ast_mut().lines.as_mut_slice()[input.index] + else { + unreachable!() + }; + utterance + .dependent_tiers + .push(DependentTier::Pho(PhoTier::from_tokens(PhoTierType::Pho, result.ipa)).into()); + } + revised.validate_stage_output(Task::Phonetic)?; + *chat = revised; + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::proto::phonetic::PhoneticResult; + use crate::utils::SourceId; + + fn chat(content: &str) -> Chat { + let linkage = if content.contains('\u{15}') { + "audio" + } else { + "audio, unlinked" + }; + Chat::parse(&format!("@UTF8\n@Begin\n@Languages:\teng\n@Participants:\tPAR Participant\n@ID:\teng|test|PAR|||||Participant|||\n@Media:\tsample, {linkage}\n{content}@End\n"), SourceId::try_new("test.cha").unwrap()).unwrap() + } + + #[test] + fn extracts_phonological_positions_including_retraces_and_pauses() { + let source = chat("*PAR:\t [/] the (.) dog . \u{15}0_1000\u{15}\n"); + let utterances = extract_utterances(&source).unwrap(); + let texts: Vec<_> = utterances[0] + .units + .iter() + .map(|unit| unit.text.as_str()) + .collect(); + assert_eq!(texts, ["the", "cat", "the", "(.)", "dog"]); + assert!(utterances[0].units[3].pause); + } + + #[test] + fn preserves_existing_pho_without_requiring_timing() { + let source = chat("*PAR:\tcat .\n%pho:\tkæt\n"); + assert!(extract_utterances(&source).unwrap().is_empty()); + } + + #[test] + fn phonological_groups_and_replacements_use_original_spoken_positions() { + let source = chat("*PAR:\t‹the cat› dog [: dogs] &-um &+ca . \u{15}0_1000\u{15}\n"); + let utterances = extract_utterances(&source).unwrap(); + let texts: Vec<_> = utterances[0] + .units + .iter() + .map(|unit| unit.text.as_str()) + .collect(); + assert_eq!(texts, ["‹the cat›", "dog", "&-um", "&+ca"]); + } + + #[test] + fn validates_all_results_before_mutating_chat() { + let mut source = + chat("*PAR:\tcat . \u{15}0_1000\u{15}\n*PAR:\tdog . \u{15}1000_2000\u{15}\n"); + let before = source.to_chat(); + let inputs = extract_utterances(&source).unwrap(); + let output = PhoneticOutput { + source_id: source.source_id().clone(), + utterances: vec![ + PhoneticResult { + index: inputs[0].index, + ipa: vec!["kæt".into()], + }, + PhoneticResult { + index: inputs[1].index, + ipa: vec![], + }, + ], + }; + assert!(inject_tiers(&mut source, &inputs, output).is_err()); + assert_eq!(source.to_chat(), before); + } + + #[test] + fn inserts_roundtrippable_ipa_without_changing_main_tier() { + let mut source = chat("*PAR:\tcat . \u{15}0_1000\u{15}\n"); + let inputs = extract_utterances(&source).unwrap(); + let output = PhoneticOutput { + source_id: source.source_id().clone(), + utterances: vec![PhoneticResult { + index: inputs[0].index, + ipa: vec!["kʰæ̃t".into()], + }], + }; + inject_tiers(&mut source, &inputs, output).unwrap(); + let serialized = source.to_chat(); + assert!(serialized.contains("%pho:\tkʰæ̃t")); + assert!(serialized.contains("*PAR:\tcat .")); + Chat::parse(&serialized, source.source_id().clone()).unwrap(); + } +} diff --git a/crates/batchalign/batchalign-engine/src/backend_impl.rs b/crates/batchalign/batchalign-engine/src/backend_impl.rs index 63881a0c..0bdc8cd3 100644 --- a/crates/batchalign/batchalign-engine/src/backend_impl.rs +++ b/crates/batchalign/batchalign-engine/src/backend_impl.rs @@ -263,6 +263,7 @@ fn call_py_backend( TaskInput::Ai(_) => "Ai", TaskInput::Asr(_) => "Asr", TaskInput::Fa(_) => "Fa", + TaskInput::Phonetic(_) => "Phonetic", TaskInput::Speaker(_) => "Speaker", TaskInput::UtSeg(_) => "UtSeg", TaskInput::Utr(_) => "Utr", diff --git a/crates/batchalign/batchalign-engine/src/dp_py.rs b/crates/batchalign/batchalign-engine/src/dp_py.rs index 65b077ce..b8506ed8 100644 --- a/crates/batchalign/batchalign-engine/src/dp_py.rs +++ b/crates/batchalign/batchalign-engine/src/dp_py.rs @@ -1,9 +1,9 @@ //! PyO3 binding for the centralized Rust Hirschberg DP aligner. //! -//! Exposes `batchalign._core.dp_align(payload, reference)` so the Python -//! morphosyntax pipeline can stop maintaining its own duplicate aligner -//! (`python/batchalign/backends/morphosyntax/ud/dp.py`, 224 LOC) and call -//! the Rust implementation directly (`batchalign_core::alignment`). +//! Exposes `batchalign._core.dp_align(payload, reference)` through the shared +//! Python wrapper `batchalign.utils.dp`. Morphosyntax, forced alignment, and +//! phonetic projection call the same Rust implementation +//! (`batchalign_core::alignment`). //! //! Match semantics are exact equality on the supplied strings (the Python //! side normalizes ahead of time when it wants case-insensitive behavior). @@ -22,8 +22,7 @@ use batchalign_core::alignment::{AlignResult, MatchMode, align}; /// - `{"type": "extra_payload", "key": str, "payload_idx": int}` /// - `{"type": "extra_reference", "key": str, "reference_idx": int}` /// -/// The Python caller maps these back to its `Match` / `Extra` dataclasses -/// during the deletion of `python/.../ud/dp.py`. +/// `batchalign.utils.dp` maps these back to its `Match` / `Extra` dataclasses. #[pyfunction] #[pyo3(signature = (payload, reference, case_insensitive=false))] pub fn dp_align( diff --git a/python/batchalign/__init__.py b/python/batchalign/__init__.py index ca17e85b..6488b3d8 100644 --- a/python/batchalign/__init__.py +++ b/python/batchalign/__init__.py @@ -48,6 +48,8 @@ "AI", "ASR", "FA", + "Phonetic", + "PhoneticXeusBackend", "Speaker", "UtSeg", "Morphosyntax", @@ -142,6 +144,8 @@ def __dir__() -> list[str]: AI, ASR, FA, + Phonetic, + PhoneticXeusBackend, Speaker, UtSeg, Morphosyntax, @@ -197,6 +201,8 @@ def __dir__() -> list[str]: "AI", "ASR", "FA", + "Phonetic", + "PhoneticXeusBackend", "Speaker", "UtSeg", "Morphosyntax", diff --git a/python/batchalign/_core/proto.py b/python/batchalign/_core/proto.py index 505c87a3..0a929924 100644 --- a/python/batchalign/_core/proto.py +++ b/python/batchalign/_core/proto.py @@ -48,6 +48,12 @@ # Forced alignment FaInput, FaOutput, + # Phonetic transcription + PhoneticInput, + PhoneticOutput, + PhoneticUtterance, + PhoneticUnit, + PhoneticResult, # Speaker diarization Diarization, DiarizationSegment, @@ -115,6 +121,7 @@ "Ai": AiInput, "Asr": AsrInput, "Fa": FaInput, + "Phonetic": PhoneticInput, "Speaker": SpeakerInput, "UtSeg": UtSegInput, # UTR's wire payload is byte-identical to AsrInput (Rust-side @@ -134,6 +141,7 @@ AiOutput: "Ai", AsrOutput: "Asr", FaOutput: "Fa", + PhoneticOutput: "Phonetic", SpeakerOutput: "Speaker", UtSegOutput: "UtSeg", MorphosyntaxOutput: "Morphosyntax", diff --git a/python/batchalign/backends/__init__.py b/python/batchalign/backends/__init__.py index 0236d84a..371bbce5 100644 --- a/python/batchalign/backends/__init__.py +++ b/python/batchalign/backends/__init__.py @@ -17,6 +17,7 @@ AI, ASR, FA, + Phonetic, Speaker, UtSeg, Morphosyntax, @@ -24,6 +25,7 @@ Coref, declared_tasks, ) +from batchalign.backends.phonetic import PhoneticXeusBackend from batchalign.backends.ai import DspyAIBackend from batchalign.backends.asr import ( AliyunAsrBackend, @@ -64,6 +66,8 @@ "AI", "ASR", "FA", + "Phonetic", + "PhoneticXeusBackend", "Speaker", "UtSeg", "Morphosyntax", diff --git a/python/batchalign/backends/base.py b/python/batchalign/backends/base.py index 654c6199..6df7c8e1 100644 --- a/python/batchalign/backends/base.py +++ b/python/batchalign/backends/base.py @@ -47,6 +47,7 @@ class Task(str, Enum): # type: ignore[no-redef] Ai = "Ai" Asr = "Asr" Fa = "Fa" + Phonetic = "Phonetic" Speaker = "Speaker" UtSeg = "UtSeg" Utr = "Utr" @@ -158,6 +159,10 @@ class FA(Backend): """Marker: this backend handles `Task.Fa` (forced alignment) inputs.""" +class Phonetic(Backend): + """Marker: acoustic phonetic transcription (`Task.Phonetic`).""" + + class Speaker(Backend): """Marker: this backend handles `Task.Speaker` (diarization) inputs.""" @@ -202,6 +207,7 @@ class Coref(Backend): AI: Task.Ai, ASR: Task.Asr, FA: Task.Fa, + Phonetic: Task.Phonetic, Speaker: Task.Speaker, UtSeg: Task.UtSeg, UTR: Task.Utr, @@ -226,6 +232,7 @@ def declared_tasks(backend: Backend) -> list[Task]: "AI", "ASR", "FA", + "Phonetic", "Speaker", "UtSeg", "Morphosyntax", diff --git a/python/batchalign/backends/fa/wav2vec2.py b/python/batchalign/backends/fa/wav2vec2.py index 09341847..69350604 100644 --- a/python/batchalign/backends/fa/wav2vec2.py +++ b/python/batchalign/backends/fa/wav2vec2.py @@ -18,7 +18,7 @@ next item is untimed — e.g. the terminal punctuation after the last word), then bound the span by the utterance window; drop impossible spans. -The DP aligner is BA2's (`backends/morphosyntax/ud/dp.py`, copied verbatim). +Sequence remapping uses `batchalign.utils.dp`, the shared Rust DP wrapper. """ from __future__ import annotations @@ -27,7 +27,7 @@ from typing import Any from batchalign.backends.base import FA, BatchPolicy -from batchalign.backends.morphosyntax.ud.dp import ( +from batchalign.utils.dp import ( Match, PayloadTarget, ReferenceTarget, diff --git a/python/batchalign/backends/fa/whisper_fa.py b/python/batchalign/backends/fa/whisper_fa.py index dae93de2..42055cbd 100644 --- a/python/batchalign/backends/fa/whisper_fa.py +++ b/python/batchalign/backends/fa/whisper_fa.py @@ -19,7 +19,7 @@ 4. Post-correct: bump each word's end to the next word's start, bound by the utterance window; drop impossible spans. -The DP aligner is BA2's (`backends/morphosyntax/ud/dp.py`, copied verbatim). +Sequence remapping uses `batchalign.utils.dp`, the shared Rust DP wrapper. Default model is `openai/whisper-large-v2` (BA2's default), loaded with `attn_implementation="eager"` so cross-attentions are available. """ @@ -29,7 +29,7 @@ from typing import Any from batchalign.backends.base import FA, BatchPolicy -from batchalign.backends.morphosyntax.ud.dp import ( +from batchalign.utils.dp import ( Match, PayloadTarget, ReferenceTarget, diff --git a/python/batchalign/backends/morphosyntax/ud/tokenize.py b/python/batchalign/backends/morphosyntax/ud/tokenize.py index 8a7b5ecf..0a663181 100644 --- a/python/batchalign/backends/morphosyntax/ud/tokenize.py +++ b/python/batchalign/backends/morphosyntax/ud/tokenize.py @@ -8,7 +8,7 @@ Port of `batchalign2/batchalign/pipelines/morphosyntax/ud.py`: `tokenizer_processor`, `conform`, `matches`, `matches_in`, `front_matches`, -`adlist_postprocessor`. The DP char-aligner lives in `dp.py` (copied verbatim). +`adlist_postprocessor`. Character alignment uses the shared `batchalign.utils.dp` wrapper. A Stanza `tokenize_postprocessor` receives a list of sentences, each a list of tokens (a token is a `str`, or a `(text, is_mwt)` tuple). We rewrite each @@ -21,7 +21,7 @@ import re from itertools import groupby -from .dp import PayloadTarget, ReferenceTarget, align +from batchalign.utils.dp import PayloadTarget, ReferenceTarget, align from .it.workarounds import NATIVE_MWT_SURFACES diff --git a/python/batchalign/backends/phonetic/__init__.py b/python/batchalign/backends/phonetic/__init__.py new file mode 100644 index 00000000..27a5ac56 --- /dev/null +++ b/python/batchalign/backends/phonetic/__init__.py @@ -0,0 +1,5 @@ +"""Acoustic phonetic transcription backends.""" + +from .xeus import PhoneticXeusBackend + +__all__ = ["PhoneticXeusBackend"] diff --git a/python/batchalign/backends/phonetic/utils/__init__.py b/python/batchalign/backends/phonetic/utils/__init__.py new file mode 100644 index 00000000..1c90d9d5 --- /dev/null +++ b/python/batchalign/backends/phonetic/utils/__init__.py @@ -0,0 +1,6 @@ +"""Phonetic reference generation and acoustic-phone projection. + +Use ``pronunciation.Pronunciations`` to obtain reference IPA for CHAT units, +then ``projection.project_phones`` to group observed phones by those units. +The references determine boundaries; output IPA always comes from the audio. +""" diff --git a/python/batchalign/backends/phonetic/utils/inference.py b/python/batchalign/backends/phonetic/utils/inference.py new file mode 100644 index 00000000..195969d1 --- /dev/null +++ b/python/batchalign/backends/phonetic/utils/inference.py @@ -0,0 +1,75 @@ +"""Windowing and length-aware inference for the pinned PhoneticXeus model. + +``group_utterances(utterances)`` returns lists of original utterance positions, +using Whisper FA's approximately 20-second span limit. A long utterance stays +whole; a gap over two seconds or backwards timing starts a new group. + +``transcribe_batch(model, waveforms)`` accepts unpadded, mono 16 kHz tensors and +returns one list of original phone tokens per waveform, in input order. It +normalizes each waveform independently, pads the batch, passes real lengths to +the encoder, and discards padded frames before greedy CTC decoding. The model +must be in eval mode. Calls must be serialized: the pinned frontend's global +normalization is temporarily disabled in favor of per-waveform normalization. +The encoder masks attention, but its convolution branches do not mask padding; +batching can therefore change some phone decisions near a window's edge. +""" + +from __future__ import annotations + +from typing import Any + + +def group_utterances(utterances: list[Any]) -> list[list[int]]: + groups: list[list[int]] = [] + for index, utterance in enumerate(utterances): + if groups: + current = groups[-1] + first = utterances[current[0]] + previous = utterances[current[-1]] + if ( + utterance.end_ms - first.start_ms <= 20_000 + and utterance.start_ms - previous.end_ms <= 2_000 + and utterance.start_ms >= previous.start_ms + and utterance.end_ms >= previous.end_ms + ): + current.append(index) + continue + groups.append([index]) + return groups + + +def transcribe_batch(model: Any, waveforms: list[Any]) -> list[list[str]]: + import torch + from torch.nn import functional as F + from torch.nn.utils.rnn import pad_sequence + + if not waveforms: + return [] + core = model.model + frontend = core.frontend + normalize = frontend.normalize_audio + with torch.inference_mode(): + waves = [wave.to(device=model.device, dtype=model.dtype) for wave in waveforms] + # Upstream F.layer_norm(x, x.shape) mixes rows and includes padding. + # Match single-utterance normalization before introducing either. + if normalize: + waves = [F.layer_norm(wave, wave.shape) for wave in waves] + lengths = torch.tensor([len(wave) for wave in waves], device=model.device) + speech = pad_sequence(waves, batch_first=True) + frontend.normalize_audio = False + try: + encoded, frame_lengths = core.encode(speech, lengths) + finally: + frontend.normalize_audio = normalize + if isinstance(encoded, tuple): + encoded = encoded[0] + ids = core.ctc.ctc_lo(encoded).argmax(dim=-1) + outputs = [] + for row, length in zip(ids, frame_lengths): + collapsed = row[:int(length)].unique_consecutive().tolist() + tokens = [core.token_list[index] for index in collapsed if index != core.blank_id] + outputs.append([ + token for token in tokens + if token and not (token.startswith("<") and token.endswith(">")) + ]) + return outputs diff --git a/python/batchalign/backends/phonetic/utils/piper.py b/python/batchalign/backends/phonetic/utils/piper.py new file mode 100644 index 00000000..a0e5548c --- /dev/null +++ b/python/batchalign/backends/phonetic/utils/piper.py @@ -0,0 +1,79 @@ +"""Convert Piper Plus 0.2.0's language-specific tokens to reference IPA. + +``piper_ipa(tokens, language)`` takes ``phonemizer.phonemize(text)`` output and +its Piper language code, e.g. ``piper_ipa(["n", "i", "tone3"], "zh")`` returns +``"ni˨˩˦"``. Most tokens are already IPA. Mandarin tone labels and Japanese +OpenJTalk labels need conversion; Japanese prosody/boundary markers are omitted. +These are reference pronunciations only, never replacements for acoustic IPA. +""" + +__all__ = ["piper_ipa", "prepare_piper"] + +_JAPANESE = { + "N": "ɴ", "N_m": "m", "N_n": "n", "N_ng": "ŋ", "N_uvular": "ɴ", + "ch": "tɕ", "sh": "ɕ", "ts": "ts", "j": "dʑ", "y": "j", + "r": "ɾ", "f": "ɸ", "u": "ɯ", "I": "i̥", "U": "ɯ̥", "g": "ɡ", + "ky": "kʲ", "gy": "ɡʲ", "kw": "kʷ", "gw": "ɡʷ", "ty": "tʲ", + "dy": "dʲ", "py": "pʲ", "by": "bʲ", "zy": "ʑ", "hy": "ç", + "ny": "nʲ", "my": "mʲ", "ry": "ɾʲ", +} +_JAPANESE_MARKERS = {"_", "#", "[", "]", "^", "$", "?", "?!", "?.", "?~"} +_TONES = {"tone1": "˥", "tone2": "˧˥", "tone3": "˨˩˦", "tone4": "˥˩", "tone5": ""} + + +def prepare_piper(language: str) -> None: + """Download missing NLTK resources before English/Korean G2P first use. + + Honor NLTK's configured data directory (including ``NLTK_DATA``). g2p-en + checks the legacy tagger at import, while current NLTK uses the English + JSON tagger at runtime; both must be available. Other languages need no + NLTK setup. Download failures propagate instead of selecting Epitran. + """ + if language not in {"en", "ko"}: + return + if language == "ko": + from importlib.util import find_spec + import sys + + # g2pk2 otherwise tries to run pip itself at inference time. + module = "eunjeon" if sys.platform == "win32" else "mecab" + if find_spec(module) is None: + raise ImportError("Korean G2P requires the complete batchalign[phonetic] extra") + import nltk + + resources = [("cmudict", "corpora/cmudict.zip")] + if language == "en": + resources += [ + ("averaged_perceptron_tagger", "taggers/averaged_perceptron_tagger.zip"), + ("averaged_perceptron_tagger_eng", "taggers/averaged_perceptron_tagger_eng/"), + ] + for package, resource in resources: + try: + nltk.data.find(resource) + except LookupError: + nltk.download( + package, download_dir=nltk.data.path[0], quiet=True, raise_on_error=True + ) + + +def piper_ipa(tokens: list[str], language: str) -> str: + """Return IPA, resolving Piper labels before shared DP comparison. + + Japanese ``cl`` (geminate closure) repeats the following consonant. + Mandarin neutral tone has no fixed pitch contour and is left unmarked. + Raises ``ValueError`` for an unresolved closure or unknown label. + """ + if language == "ja": + phones = [_JAPANESE.get(t, t) for t in tokens if t not in _JAPANESE_MARKERS] + for index, phone in enumerate(phones): + if phone == "cl": + if index + 1 == len(phones) or phones[index + 1] == "cl": + raise ValueError("Piper Japanese returned an unresolved geminate closure") + phones[index] = phones[index + 1][0] + elif language == "zh": + phones = [_TONES.get(t, t) for t in tokens] + else: + phones = [{"rr": "r", "y_vowel": "y"}.get(t, t) for t in tokens] + if any("_" in phone or phone.startswith("tone") for phone in phones): + raise ValueError("Piper returned an unknown phoneme label") + return "".join(phones) diff --git a/python/batchalign/backends/phonetic/utils/projection.py b/python/batchalign/backends/phonetic/utils/projection.py new file mode 100644 index 00000000..09b67d59 --- /dev/null +++ b/python/batchalign/backends/phonetic/utils/projection.py @@ -0,0 +1,128 @@ +"""Group observed IPA phones into transcript units using reference IPA. + +Use ``project_phones(phones, pronunciations)``:: + + from batchalign.backends.phonetic.utils.projection import project_phones + + project_phones(["ð", "ə", "t", "æ", "t"], ["ðə", "kæt"]) + # ["ðə", "tæt"]: the observed substitution survives unchanged. + +Pass the acoustic model's original phone tokens and one reference IPA string +per spoken CHAT unit, in transcript order. References can come from +``pronunciation.Pronunciations`` or the caller. Handle pauses separately; +do not pass them as spoken units. No G2P or audio model is loaded here. +""" + +from __future__ import annotations + +import unicodedata +from collections import Counter + +from batchalign.utils.dp import ( + ExtraType, + Match, + PayloadTarget, + ReferenceTarget, + align, +) + + +# CHAT's single-character marker for an uncoded item on the %pho tier. +# https://talkbank.org/0info/manuals/CHAT.html (Special Form Markers) +UNRESOLVED = "…" + +__all__ = ["project_phones", "comparison_symbols", "UNRESOLVED"] + + +def comparison_symbols(ipa: str) -> list[str]: + """IPA segments for DP, retaining contrastive diacritics and modifiers. + + Canonical Unicode equivalents compare equally. Stress and syllable/word + separators do not own phones; tied affricates remain a single segment. + There are no language-specific vowel or rhotic equivalences. + """ + symbols: list[str] = [] + tied = False + for ch in unicodedata.normalize("NFD", ipa.replace("g", "ɡ")): + if ch.isspace() or ch in "ˈˌ.‿": + continue + if symbols and (unicodedata.combining(ch) or ch in "ːˑʰʷʲⁿˡ˞ˠˤʼ" or tied): + symbols[-1] += ch + else: + symbols.append(ch) + tied = ch in "\u0361\u035c" + return symbols + + +def project_phones(phones: list[str], pronunciations: list[str]) -> list[str]: + """Assign every original phone once using BA's existing edit alignment. + + Args: + phones: Observed IPA tokens in audio order. A token can contain + multiple symbols or diacritics; it is never split in the output. + pronunciations: Reference IPA, one string per spoken transcript unit. + + Returns: + One observed IPA string per reference unit, or ``UNRESOLVED`` (``…``) + when no phones align to it. An empty acoustic sequence produces only + these placeholders. Removing placeholders and concatenating the result + equals ``"".join(phones)`` exactly. Inputs are not mutated. + + Raises: + ValueError: A reference has no comparison symbols, a token cannot be + assigned, or assignments cross unit boundaries. + + Within an edit run, pair substitutions in order. Remaining insertions + attach to the preceding reference unit (the following unit at the start). + Multi-symbol phones use majority ownership, with ties going to the left. + A unit receiving no acoustic phones is marked unresolved, never filled + from G2P. The placeholder means uncoded, not an observed pause. + """ + payload = [ + PayloadTarget(symbol, index) + for index, phone in enumerate(phones) + for symbol in comparison_symbols(phone) + ] + reference_symbols = [comparison_symbols(ipa) for ipa in pronunciations] + if not reference_symbols or any(not symbols for symbols in reference_symbols): + raise ValueError("Cannot project empty pronunciation sequences") + reference = [ + ReferenceTarget(symbol, index) + for index, symbols in enumerate(reference_symbols) + for symbol in symbols + ] + edits = align(payload, reference, tqdm=False) + votes: list[list[int]] = [[] for _ in phones] + previous = reference[0].payload + offset = 0 + while offset < len(edits): + edit = edits[offset] + if isinstance(edit, Match): + votes[edit.payload].append(edit.reference_payload) + previous = edit.reference_payload + offset += 1 + continue + inserted, deleted = [], [] + while offset < len(edits) and not isinstance(edits[offset], Match): + extra = edits[offset] + (inserted if extra.extra_type == ExtraType.PAYLOAD else deleted).append( + extra.payload + ) + offset += 1 + for index, phone in enumerate(inserted): + owner = deleted[min(index, len(deleted) - 1)] if deleted else previous + votes[phone].append(owner) + if deleted: + previous = deleted[-1] + result = [""] * len(pronunciations) + previous = 0 + for phone, owners in zip(phones, votes): + if not owners: + raise ValueError(f"Phone {phone!r} has no comparison symbols") + counts = Counter(owners) + owner = min(counts, key=lambda unit: (-counts[unit], unit)) + if owner < previous: + raise ValueError("Phone projection crossed a word boundary") + result[owner] += phone + previous = owner + return [unit or UNRESOLVED for unit in result] diff --git a/python/batchalign/backends/phonetic/utils/pronunciation.py b/python/batchalign/backends/phonetic/utils/pronunciation.py new file mode 100644 index 00000000..146380db --- /dev/null +++ b/python/batchalign/backends/phonetic/utils/pronunciation.py @@ -0,0 +1,183 @@ +"""Generate reference IPA for CHAT units using Piper Plus, Epitran, and CSV overrides. + +Create one ``Pronunciations`` instance per backend and call it with each +spoken unit's text and the primary ISO 639-3 language supplied by CHAT:: + + from batchalign.backends.phonetic.utils.pronunciation import ( + Pronunciations, load_pronunciations, + ) + + pronounce = Pronunciations(load_pronunciations("pronunciations.csv")) + pronounce("gato", "spa") # reference IPA from Piper Plus + pronounce("wug", "eng") # "wʌɡ" if present in the CSV + +Each call returns one reference IPA string for the unit, including grouped +words. Feed these strings to ``projection.project_phones`` to recover unit +boundaries without replacing acoustic IPA with expected pronunciations. +Piper Plus handles its supported languages; Epitran handles the rest. +Engines load lazily and are reused. Python 3.11+ is required. Language +resources may download on first use; English needs no system G2P executable. +""" + +from __future__ import annotations + +import csv +import re +import sys +from importlib.metadata import version +from pathlib import Path +from typing import Any, Mapping + +__all__ = ["Pronunciations", "epitran_code", "load_pronunciations"] + +# Piper uses ISO 639-1; CHAT supplies ISO 639-3. Mandarin's individual +# language code has no alpha-2 equivalent in pycountry. +_PIPER_LANGUAGES = { + "eng": "en", "jpn": "ja", "cmn": "zh", "zho": "zh", "kor": "ko", + "spa": "es", "fra": "fr", "por": "pt", "swe": "sv", +} + + +def load_pronunciations(path: str | Path) -> dict[str, str]: + """Read a UTF-8 CSV containing exactly the columns ``word`` and ``ipa``. + + Example file (the header is required):: + + word,ipa + wug,wʌɡ + bonjour,bɔ̃ʒuʁ + the cat,ðəkæt + + Words may be whole CHAT phonological units. Standard CSV quoting handles + commas. Blank lines are ignored and a UTF-8 BOM is accepted. Returns a + case-insensitive word-to-IPA mapping for ``Pronunciations(overrides)``. + Raises ``ValueError`` for malformed rows, empty cells, or duplicate words; + file access errors propagate as ``OSError``. + """ + result: dict[str, str] = {} + try: + with Path(path).open(encoding="utf-8-sig", newline="") as source: + rows = csv.reader(source, strict=True) + if next(rows, None) != ["word", "ipa"]: + raise ValueError("pronunciation CSV must start with the header word,ipa") + for row in rows: + if not row: + continue + if len(row) != 2 or not all(cell.strip() for cell in row): + raise ValueError(f"CSV line {rows.line_num}: expected nonempty word,ipa") + word, ipa = (cell.strip() for cell in row) + word = word.casefold() + if word in result: + raise ValueError(f"CSV line {rows.line_num}: duplicate word {word!r}") + result[word] = ipa + except csv.Error as error: + raise ValueError(f"Invalid pronunciation CSV: {error}") from error + return result + + +def epitran_code(language: str) -> str: + """Resolve CHAT's ISO language to Epitran's language/default-script pair.""" + from langcodes import Language + from batchalign.lang import LanguageCode + + language = LanguageCode.from_str(language).alpha_3 + script = Language.get(language).maximize().script + # CHAT commonly uses the Chinese macrolanguage; Epitran names Mandarin. + language = "cmn" if language == "zho" else language + return f"{language}-{script}" + + +class Pronunciations: + """Generate comparison IPA only; the acoustic phones remain authoritative. + + The task runner passes CHAT's primary language; langcodes supplies its + default script for the Epitran fallback, e.g. ``rus`` becomes ``rus-Cyrl``. + """ + + def __init__(self, overrides: Mapping[str, str] | None = None): + if sys.version_info < (3, 11): + raise ValueError("Phonetic pronunciation generation requires Python 3.11 or newer") + if any( + not isinstance(word, str) or not word.strip() + or not isinstance(ipa, str) or not ipa.strip() + for word, ipa in (overrides or {}).items() + ): + raise ValueError( + "pronunciations must map nonempty words to nonempty IPA strings" + ) + self.overrides = { + word.casefold(): ipa for word, ipa in (overrides or {}).items() + } + self.version = ( + f"piper-{version('piper-plus-g2p')}:epitran-{version('epitran')}" + f":langcodes-{version('langcodes')}" + ) + self._piper_engines: dict[str, Any] = {} + self._epitran_engines: dict[str, Any] = {} + + def _epitran_engine(self, language: str) -> Any: + import epitran + from epitran.exceptions import DatafileError + + code = epitran_code(language) + if code not in self._epitran_engines: + try: + self._epitran_engines[code] = epitran.Epitran(code, tones=True) + except (OSError, ValueError, DatafileError) as error: + raise ValueError( + f"Cannot initialize Epitran {code!r}: {error}; check CHAT's " + "@Languages and language resources or supply IPA pronunciation overrides" + ) from error + return self._epitran_engines[code] + + def _pronounce(self, word: str, language: str) -> str: + from batchalign.lang import LanguageCode + + language = LanguageCode.from_str(language).alpha_3 + if code := _PIPER_LANGUAGES.get(language): + from piper_plus_g2p import get_phonemizer + from .piper import piper_ipa, prepare_piper + + if code not in self._piper_engines: + prepare_piper(code) + self._piper_engines[code] = get_phonemizer(code) + # A broken supported backend must surface its error, not silently + # change providers (particularly back to Flite for English). + ipa = piper_ipa(self._piper_engines[code].phonemize(word), code) + else: + engine = self._epitran_engine(language) + ipa = engine.transliterate(word) + if ipa != engine.strict_trans(word): + raise ValueError( + f"Incomplete IPA pronunciation for {word!r} in {language!r}; " + "supply IPA pronunciation overrides" + ) + if not ipa.strip(): + raise ValueError(f"Empty IPA pronunciation for {word!r} in {language!r}") + return ipa + + def __call__(self, text: str, language: str) -> str: + """Return reference IPA for a spoken CHAT unit and ISO 639-3 language. + + Whole-unit and individual-word overrides take precedence over G2P. + CHAT group/word markers are removed before lookup. Raises ``ValueError`` + for invalid language codes, unavailable language resources, incomplete + transliteration, or units without spoken words. Supply overrides for + words the selected provider cannot handle. + """ + if text.casefold() in self.overrides: + return self.overrides[text.casefold()] + # CHAT group delimiters and word-form markers carry no spoken phones. + text = re.sub(r"\[[^\]]*\]", "", text).strip("‹›<>") + words = re.sub(r"&[-+~]", "", text).replace("+", " ").split() + phones = [] + for word in words: + word = re.sub(r"^&[-+~]", "", word).split("@", 1)[0] + word = word.replace("(", "").replace(")", "").casefold() + if word in self.overrides: + phones.append(self.overrides[word]) + continue + phones.append(self._pronounce(word, language)) + if not phones: + raise ValueError(f"No spoken words in phonological unit {text!r}") + return "".join(phones) diff --git a/python/batchalign/backends/phonetic/xeus.py b/python/batchalign/backends/phonetic/xeus.py new file mode 100644 index 00000000..cd69ae05 --- /dev/null +++ b/python/batchalign/backends/phonetic/xeus.py @@ -0,0 +1,157 @@ +"""PhoneticXeus inference with word ownership projected through phone DP.""" + +from __future__ import annotations + +import hashlib +import json +import logging +import threading +from typing import Any, Mapping + +from batchalign.backends.base import BatchPolicy, Phonetic +from .utils.inference import group_utterances, transcribe_batch +from .utils.pronunciation import Pronunciations +from .utils.projection import UNRESOLVED, project_phones + +logger = logging.getLogger(__name__) + +MODEL = "changelinglab/PhoneticXeus" +REVISION = "3a8d860fa68f8936ceb4196651221215bab9dae4" + + +class PhoneticXeusBackend(Phonetic): + """Observed IPA from audio; reference pronunciations determine grouping only.""" + + def __init__( + self, + *, + model: str = MODEL, + revision: str = REVISION, + device: str | None = None, + pronunciations: Mapping[str, str] | None = None, + batch_size: int | None = None, + ): + if len(revision) != 40 or any(ch not in "0123456789abcdef" for ch in revision): + raise ValueError( + "PhoneticXeus revision must be an immutable 40-character commit SHA" + ) + if batch_size is not None and batch_size < 1: + raise ValueError("batch_size must be positive") + self.batch_size = batch_size + self.model_id, self.revision, self.device = model, revision, device + self.pronunciations = Pronunciations(pronunciations) + # Textual replaces stderr with a stream whose fileno() is -1. Hugging + # Face's first progress bar creates a multiprocessing resource tracker, + # which cannot inherit that descriptor. Initialize its lock here, before + # the dashboard starts and inference moves to an engine worker thread. + from huggingface_hub.utils import tqdm + + tqdm.get_lock() + self._model = None + self._lock = threading.Lock() + + @property + def name(self) -> str: + overrides = json.dumps( + self.pronunciations.overrides, sort_keys=True, ensure_ascii=False + ) + digest = hashlib.sha256(overrides.encode()).hexdigest() + return f"phoneticxeus:{self.model_id}:{self.revision}:{self.pronunciations.version}:{digest}:{self.device}:{self.batch_size}:v6" + + @property + def batch_policy(self) -> BatchPolicy: + return BatchPolicy.one() + + def call( + self, batch: list[Any], *, progress: Any = None, **_kwargs: Any + ) -> list[Any]: + import numpy as np + import torch + import torchaudio.functional as audio_ops + from huggingface_hub import snapshot_download + from transformers import AutoModel + from batchalign._core.proto import PhoneticOutput, PhoneticResult + + outputs = [] + device = self.device or ("cuda" if torch.cuda.is_available() else "cpu") + batch_size = self.batch_size or (2 if str(device).startswith("cuda") else 1) + with self._lock: + for item in batch: + sample_rate = int(item.audio.sample_rate) + if sample_rate <= 0: + raise ValueError( + "PhoneticXeus requires a positive audio sample rate" + ) + waveform = np.frombuffer(item.audio.pcm_f32le, dtype=" None: + @app.command() + def phonetic( + ctx: typer.Context, + paths: list[Path] | None = typer.Argument( + None, + exists=True, + help="Timed CHAT files or directories with matching audio.", + ), + input_list: Path | None = typer.Option( + None, "--input-list", "--file-list", "-i", exists=True, dir_okay=False + ), + out: Path | None = typer.Option( + None, "--out", "-o", help="Output directory; defaults to writing in place." + ), + pronunciations: Path | None = typer.Option( + None, + "--pronunciations", + exists=True, + dir_okay=False, + help="UTF-8 CSV overrides with header word,ipa and one word or CHAT unit " + "per row. Example row: wug,wʌɡ. Overrides the generated pronunciation for that unit.", + ), + force_cpu: bool = typer.Option(False, "--force-cpu", help="Use CPU inference."), + batch_size: int | None = typer.Option( + None, "--batch-size", min=1, + help="Audio windows per padded model batch (default: CPU 1, CUDA 2). " + "Lower this to reduce memory use.", + ), + ) -> None: + """Add observed IPA to `%pho`, preserving existing phonetic tiers. + + Requires utterance timing bullets (run utr first if absent). + Adjacent utterances share approximately 20-second inference windows. + Reference IPA uses Piper Plus for supported CHAT languages and Epitran + otherwise. Requires Python 3.11+; English needs no system G2P executable. + + Override example: --pronunciations pronunciations.csv + + CSV contents (header required): + \b + word,ipa + wug,wʌɡ + bonjour,bɔ̃ʒuʁ + """ + import batchalign as ba + from batchalign.backends.phonetic.utils.pronunciation import load_pronunciations + + selection = resolve_inputs(paths, input_list, CHAT_EXTENSIONS) + opts = cli_options(ctx) + overrides = None + if pronunciations is not None: + try: + overrides = load_pronunciations(pronunciations) + except (ValueError, OSError) as error: + raise typer.BadParameter( + str(error), param_hint="--pronunciations" + ) from error + with Interface.open( + command="phonetic", + params={"engine": "phoneticxeus"}, + output=out, + verbosity=opts.verbosity, + plain=opts.plain, + quiet=opts.quiet, + ) as ui: + backend = ba.PhoneticXeusBackend( + device=inference_device(force_cpu=force_cpu, allow_mps=False), + pronunciations=overrides, + batch_size=batch_size, + ) + pipeline = ba.recipes.phonetic( + phonetic_backend=backend, workers=opts.parallel + ) + inputs, root = collect_chat_inputs(selection) + for item in inputs: + ui.push(Task.from_input(item)) + list( + ui.run_pipeline( + pipeline, + inputs, + on_outcome=lambda outcome: write_outcome(outcome, root, out), + ) + ) + raise typer.Exit(code=ui.exit_code) diff --git a/python/batchalign/recipes.py b/python/batchalign/recipes.py index aa386fc4..8b9a6280 100644 --- a/python/batchalign/recipes.py +++ b/python/batchalign/recipes.py @@ -127,6 +127,19 @@ def utr(*, utr_backend: Any, **opts: Any) -> Any: return Pipeline(tasks=[Task.Utr], backends=[utr_backend], **opts) +def phonetic( + *, phonetic_backend: Any, utr_backend: Any | None = None, **opts: Any +) -> Any: + """Add observed IPA to `%pho`; optionally recover utterance timing first.""" + Task, Pipeline = _core() + tasks = [Task.Phonetic] + backends = [phonetic_backend] + if utr_backend is not None: + tasks.insert(0, Task.Utr) + backends.insert(0, utr_backend) + return Pipeline(tasks=tasks, backends=backends, **opts) + + def morphotag(*, stanza_backend: Any, **opts: Any) -> Any: """Morphosyntax tagging via Stanza (UD `%mor` / `%gra`).""" Task, Pipeline = _core() @@ -213,6 +226,7 @@ def compare( "align", "utr", "morphotag", + "phonetic", "translate", "ai", "coref", diff --git a/python/batchalign/tests/test_dp_shim.py b/python/batchalign/tests/test_dp_shim.py index 0351806f..4b597b5c 100644 --- a/python/batchalign/tests/test_dp_shim.py +++ b/python/batchalign/tests/test_dp_shim.py @@ -7,7 +7,7 @@ from __future__ import annotations -from batchalign.backends.morphosyntax.ud.dp import ( +from batchalign.utils.dp import ( Extra, ExtraType, Match, diff --git a/python/batchalign/tests/test_phonetic.py b/python/batchalign/tests/test_phonetic.py new file mode 100644 index 00000000..aa859c7e --- /dev/null +++ b/python/batchalign/tests/test_phonetic.py @@ -0,0 +1,510 @@ +"""Phone ownership, multilingual IPA references, and the real CLI/runner seam.""" + +from __future__ import annotations + +import wave +import base64 +from types import SimpleNamespace + +import pytest +from typer.testing import CliRunner + +from batchalign.backends.phonetic.utils.pronunciation import ( + Pronunciations, epitran_code, load_pronunciations, +) +from batchalign.backends.phonetic.utils.projection import comparison_symbols, project_phones +from batchalign.backends.phonetic.utils.piper import piper_ipa +from batchalign.cli import app + + +@pytest.mark.parametrize( + "phones,reference,expected", + [ + (["ð", "ə", "k", "æ", "t"], ["ðə", "kæt"], ["ðə", "kæt"]), + # A substitution must retain the observed t, not canonical k. + (["ð", "ə", "t", "æ", "t"], ["ðə", "kæt"], ["ðə", "tæt"]), + (["ð", "ə", "kʰ", "æ̃", "t", "s"], ["ðə", "kæt"], ["ðə", "kʰæ̃ts"]), + (["ə", "b", "b", "ə"], ["əb", "bə"], ["əb", "bə"]), + (["aɪ", "s", "iː"], ["aɪ", "si"], ["aɪ", "siː"]), + (["s", "ə", "k", "æ", "t"], ["ðə", "kæt"], ["sə", "kæt"]), + ], +) +def test_projection_preserves_observed_phones(phones, reference, expected): + assert project_phones(phones, reference) == expected + assert "".join(expected) == "".join(phones) + + +@pytest.mark.parametrize( + "phones,reference,expected", + [ + (["k", "æ", "t"], ["ðə", "kæt"], ["…", "kæt"]), + (["k", "æ", "t"], ["kæt", "ðə"], ["kæt", "…"]), + (["k", "æ", "t", "d", "ɒ", "ɡ"], ["kæt", "ə", "dɒɡ"], ["kæt", "…", "dɒɡ"]), + ([], ["kæt", "dɒɡ"], ["…", "…"]), + # Actual Xeus output for "just to test batch line", 5305–6555 ms. + (["t", "ʃ", "ɪ", "s", "t", "æ", "z", "æ", "s", "p", "æ", "ʃ", "ə", "l", "a", "ɪ̃", "n"], + ["dʒˈʌst", "tuː", "tˈɛst", "bˈætʃ", "lˈaɪn"], + ["…", "tʃ", "ɪst", "æzæspæʃə", "laɪ̃n"]), + ], +) +def test_projection_marks_unresolved_units(phones, reference, expected): + assert project_phones(phones, reference) == expected + assert "".join(unit for unit in expected if unit != "…") == "".join(phones) + + +@pytest.mark.parametrize("phones,reference", [(["k"], []), (["k"], [""]), ([""], ["k"])]) +def test_projection_still_rejects_invalid_input(phones, reference): + with pytest.raises(ValueError): + project_phones(phones, reference) + + +def test_comparison_normalization_does_not_split_combining_marks(): + assert comparison_symbols("ˈkʰæ̃tː") == ["kʰ", "æ̃", "tː"] + assert comparison_symbols("ã") == comparison_symbols("a\u0303") + assert comparison_symbols("t͡ʃ") == ["t͡ʃ"] + assert comparison_symbols("rɹɚɝʌəɐ") == list("rɹɚɝʌəɐ") + assert comparison_symbols("a˥a˩") == list("a˥a˩") + + +@pytest.mark.parametrize( + "language,word,ipa", + [("spa", "gato", "ɡato"), ("tur", "göz", "ɡœz"), ("fra", "chat", "ʃa")], +) +def test_multilingual_ipa_references(language, word, ipa): + reference = Pronunciations()(word, language) + assert comparison_symbols(reference) == comparison_symbols(ipa) + # The same DP groups directly against generated IPA for each language. + phones = comparison_symbols(ipa) * 2 + assert project_phones(phones, [reference, reference]) == [ipa, ipa] + + +def test_automatic_non_latin_script(): + assert Pronunciations()("кот", "rus") == "kot" + + +@pytest.mark.parametrize( + "language,code", + [("fra", "fra-Latn"), ("rus", "rus-Cyrl"), ("hin", "hin-Deva"), + ("ara", "ara-Arab"), ("cmn", "cmn-Hans"), ("zho", "cmn-Hans"), + ("yue", "yue-Hant"), ("jpn", "jpn-Jpan")], +) +def test_language_selects_epitran_script(language, code): + assert epitran_code(language) == code + + +def test_lookup_overrides_and_chat_markers(): + lookup = Pronunciations({"wug": "wʌɡ", "bonjour": "bɔ̃ʒuʁ"}) + assert comparison_symbols(lookup("&-gato", "spa")) == list("ɡato") + assert comparison_symbols(lookup("‹gato gato›", "spa")) == list("ɡatoɡato") + assert lookup("wug", "eng") == "wʌɡ" + assert lookup("bonjour", "fra") == "bɔ̃ʒuʁ" + with pytest.raises(ValueError, match="@Languages"): + lookup("word", "ell") # A valid CHAT language without an Epitran map. + with pytest.raises(ValueError, match="Incomplete IPA"): + lookup("göz猫", "tur") + + +def test_pronunciation_csv(tmp_path): + path = tmp_path / "pronunciations.csv" + path.write_text('\ufeffword,ipa\nWug,wʌɡ\n"the cat",ðəkæt\n"a,b",ab\n\n') + overrides = load_pronunciations(path) + assert overrides == {"wug": "wʌɡ", "the cat": "ðəkæt", "a,b": "ab"} + assert Pronunciations(overrides)("WUG", "eng") == "wʌɡ" + + +@pytest.mark.parametrize( + "contents,error", + [ + ('{"wug": "wʌɡ"}', "header word,ipa"), + ("word,ipa\nwug,\n", "nonempty"), + ("word,ipa\nwug,wʌɡ,extra\n", "nonempty"), + ("word,ipa\nwug,wʌɡ\nWUG,wʊɡ\n", "duplicate"), + ('word,ipa\n"wug,wʌɡ\n', "Invalid pronunciation CSV"), + ], +) +def test_pronunciation_csv_rejects_invalid_input(tmp_path, contents, error): + path = tmp_path / "pronunciations.csv" + path.write_text(contents) + with pytest.raises(ValueError, match=error): + load_pronunciations(path) + + +def test_english_uses_piper_without_flite(monkeypatch, tmp_path): + import nltk + + monkeypatch.setattr(nltk.data, "path", [str(tmp_path)]) + monkeypatch.setattr("shutil.which", lambda _: None) + def no_epitran(*args): + pytest.fail("English must not use Epitran") + monkeypatch.setattr(Pronunciations, "_epitran_engine", no_epitran) + assert comparison_symbols(Pronunciations()("cat", "eng")) == list("kæt") + assert Pronunciations({"cat": "kæt"})("cat", "eng") == "kæt" + + +def test_piper_failure_does_not_silently_fall_back(monkeypatch): + import piper_plus_g2p + + def broken(code): + raise RuntimeError("Piper unavailable") + + monkeypatch.setattr(piper_plus_g2p, "get_phonemizer", broken) + with pytest.raises(RuntimeError, match="Piper unavailable"): + Pronunciations()("gato", "spa") + # Explicit overrides still bypass G2P entirely. + assert Pronunciations({"cat": "kæt"})("cat", "eng") == "kæt" + + +def test_unsupported_piper_language_uses_epitran(monkeypatch): + import piper_plus_g2p + + def no_piper(code): + pytest.fail("Russian must use Epitran") + + monkeypatch.setattr(piper_plus_g2p, "get_phonemizer", no_piper) + assert Pronunciations()("кот", "rus") == "kot" + + +@pytest.mark.parametrize("language", ["cmn", "zho"]) +def test_mandarin_piper_tones(language): + assert Pronunciations()("你", language) == "ni˨˩˦" + + +@pytest.mark.parametrize("language,word,expected", [("jpn", "猫", "neko"), ("kor", "가", "ka")]) +def test_piper_asian_backends(language, word, expected, monkeypatch, tmp_path): + import nltk + + monkeypatch.setattr(nltk.data, "path", [str(tmp_path)]) + assert Pronunciations()(word, language) == expected + + +@pytest.mark.parametrize( + "language,tokens,expected", + [ + ("zh", ["n", "i", "tone3", "x", "au", "tone4"], "ni˨˩˦xau˥˩"), + ("ja", ["k", "o", "[", "N_n", "n", "i", "ch", "i", "w", "a"], "konnitɕiwa"), + ("ja", ["k", "i", "cl", "t", "e", "]"], "kitte"), + ("es", ["p", "e", "rr", "o"], "pero"), + ], +) +def test_piper_labels_are_converted_to_ipa(language, tokens, expected): + assert piper_ipa(tokens, language) == expected + + +@pytest.fixture +def transcript(tmp_path): + path = tmp_path / "sample.cha" + path.write_text( + "@UTF8\n@Begin\n@Languages:\teng\n@Participants:\tPAR Participant\n" + "@ID:\teng|test|PAR|||||Participant|||\n@Media:\trecording, audio\n" + "*PAR:\tthe (.) cat . \x150_1000\x15\n@End\n", + encoding="utf-8", + ) + with wave.open(str(tmp_path / "recording.wav"), "wb") as audio: + audio.setnchannels(1) + audio.setsampwidth(2) + audio.setframerate(16000) + audio.writeframes(b"\x00\x00" * 16000) + return path + + +@pytest.fixture +def fake_backend(monkeypatch): + import batchalign as ba + from batchalign._core.proto import PhoneticOutput, PhoneticResult + + class FakePhonetic(ba.Phonetic): + calls = [] + corrupt = False + phone = "tə" + + @property + def name(self): + # Distinct cache namespace per test instance. + return f"phonetic-test-{id(self)}" + + @property + def batch_policy(self): + return ba.BatchPolicy.one() + + def call(self, batch, **kwargs): + self.calls.extend(batch) + outputs = [] + for item in batch: + utterances = [] + for utterance in item.utterances: + ipa = [ + unit.text if unit.pause else self.phone for unit in utterance.units + ] + if self.corrupt: + ipa.pop() + utterances.append(PhoneticResult(index=utterance.index, ipa=ipa)) + outputs.append( + PhoneticOutput(source_id=item.source_id, utterances=utterances) + ) + return outputs + + backend = FakePhonetic() + monkeypatch.setattr(ba, "PhoneticXeusBackend", lambda **kwargs: backend) + return backend + + +@pytest.mark.parametrize("phone", ["tə", "…"]) +@pytest.mark.parametrize("language", ["eng", "fra", "rus"]) +def test_cli_runner_writes_pho_and_preserves_existing( + transcript, fake_backend, tmp_path, language, phone +): + fake_backend.phone = phone + transcript.write_text(transcript.read_text().replace("eng", language)) + output = tmp_path / "out" + result = CliRunner().invoke( + app, ["phonetic", str(transcript), "--out", str(output)] + ) + assert result.exit_code == 0, result.output + text = (output / transcript.name).read_text() + assert f"%pho:\t{phone} (.) {phone}" in text + assert "%pho:" not in transcript.read_text() + units = fake_backend.calls[0].utterances[0].units + assert fake_backend.calls[0].language == language + assert [(unit.text, unit.pause) for unit in units] == [ + ("the", False), + ("(.)", True), + ("cat", False), + ] + # Existing tiers skip inference, even without media in the output folder. + calls = len(fake_backend.calls) + result = CliRunner().invoke(app, ["phonetic", str(output / transcript.name)]) + assert result.exit_code == 0, result.output + assert len(fake_backend.calls) == calls + + +def test_cli_failure_does_not_overwrite_source(transcript, fake_backend): + fake_backend.corrupt = True + original = transcript.read_bytes() + result = CliRunner().invoke(app, ["phonetic", str(transcript)]) + assert result.exit_code != 0 + assert transcript.read_bytes() == original + + +def test_cli_requires_utterance_timing(transcript, fake_backend): + transcript.write_text( + transcript.read_text() + .replace(" \x150_1000\x15", "") + .replace("recording, audio", "recording, audio, unlinked") + ) + result = CliRunner().invoke(app, ["phonetic", str(transcript)]) + assert result.exit_code != 0 + assert "run utr first" in result.output + assert not fake_backend.calls + + +def test_cli_help_is_lazy(): + result = CliRunner().invoke(app, ["phonetic", "--help"]) + assert result.exit_code == 0 + assert "--pronunciations" in result.output + assert "word,ipa" in result.output + assert "wug,wʌɡ" in result.output + assert "--g2p-code" not in result.output + + +def test_cli_loads_csv_overrides(transcript, fake_backend, tmp_path, monkeypatch): + import batchalign as ba + + path = tmp_path / "pronunciations.csv" + path.write_text("word,ipa\ncat,kæt\n") + supplied = {} + + def backend(**kwargs): + supplied.update(kwargs) + return fake_backend + + monkeypatch.setattr(ba, "PhoneticXeusBackend", backend) + result = CliRunner().invoke( + app, ["phonetic", str(transcript), "--pronunciations", str(path)] + ) + assert result.exit_code == 0, result.output + assert supplied["pronunciations"] == {"cat": "kæt"} + + +@pytest.mark.parametrize( + "language,word,overrides,observed,expected", + [("eng", "cat", {"cat": "kæt"}, "/kʰ/æ̃/t", "kʰæ̃t"), + ("eng", "cat", {"cat": "kæt"}, "", "…"), + ("rus", "кот", None, "/k/o/t", "kot")], +) +def test_backend_resamples_and_retains_phone_tokens( + language, word, overrides, observed, expected, caplog, monkeypatch +): + from batchalign.backends.phonetic.xeus import PhoneticXeusBackend + from batchalign._core.proto import PreparedAudio + + calls = [] + + def transcribe(model, audio_batch): + calls.extend((len(audio), 16000) for audio in audio_batch) + return [[p for p in observed.split("/") if p != ""]] + + monkeypatch.setattr("batchalign.backends.phonetic.xeus.transcribe_batch", transcribe) + + backend = PhoneticXeusBackend(pronunciations=overrides) + backend._model = object() + item = SimpleNamespace( + source_id="sample", + language=language, + audio=PreparedAudio( + pcm_f32le=base64.b64encode(b"\x00" * 22050 * 4), + sample_rate=22050, + channels=1, + frame_count=22050, + ), + utterances=[ + SimpleNamespace( + index=0, + start_ms=0, + end_ms=1000, + units=[SimpleNamespace(text=word, pause=False)], + ) + ], + ) + output = backend.call([item]) + assert calls == [(16000, 16000)] + assert output[0].utterances[0].ipa == [expected] + if expected == "…": + assert "sample: phonetic 0–1000 ms: no phones aligned to ['cat']" in caplog.text + else: + assert "no phones aligned" not in caplog.text + item.utterances[0].end_ms = 2000 + with pytest.raises(ValueError, match="outside"): + backend.call([item]) + + +def test_backend_prepares_download_lock_before_inference(monkeypatch): + """The download lock must exist before Textual redirects stderr.""" + from huggingface_hub.utils import tqdm + from batchalign.backends.phonetic.xeus import PhoneticXeusBackend + + calls = [] + get_lock = tqdm.get_lock + + def prepare_lock(): + calls.append(get_lock()) + return calls[-1] + + monkeypatch.setattr(tqdm, "get_lock", prepare_lock) + backend = PhoneticXeusBackend() + assert len(calls) == 1 + assert backend._model is None + + +def test_backend_identity_tracks_pronunciations_and_pins_revision(): + from batchalign.backends.phonetic.xeus import PhoneticXeusBackend + + default = PhoneticXeusBackend() + override = PhoneticXeusBackend(pronunciations={"cat": "kɛt"}) + assert default.name != override.name + assert override.name == PhoneticXeusBackend(pronunciations={"cat": "kɛt"}).name + assert default.name != PhoneticXeusBackend(batch_size=2).name + with pytest.raises(ValueError, match="batch_size"): + PhoneticXeusBackend(batch_size=0) + with pytest.raises(ValueError, match="immutable"): + PhoneticXeusBackend(revision="main") + + +def test_phonetic_grouping_matches_whisper_span_limit(): + from batchalign.backends.phonetic.utils.inference import group_utterances + + windows = [(0, 10000), (10000, 20000), (20000, 21000), (24001, 25000), + (23000, 24000), (25000, 50000), (50000, 51000)] + utterances = [SimpleNamespace(start_ms=start, end_ms=end) for start, end in windows] + assert group_utterances(utterances) == [[0, 1], [2], [3], [4], [5], [6]] + assert group_utterances([]) == [] + + +def test_phonetic_padded_ctc_uses_lengths_and_independent_normalization(): + import torch + from batchalign.backends.phonetic.utils.inference import transcribe_batch + + frontend = SimpleNamespace(normalize_audio=True) + calls = [] + + def encode(speech, lengths): + assert not frontend.normalize_audio + assert not torch.is_grad_enabled() + calls.append((speech.clone(), lengths.tolist())) + # A repeated phone across a blank must survive. The trailing b frames + # in row 0 are padding and must NOT become an observed phone. + ids = torch.tensor([[1, 1, 0, 1, 2, 2], [2, 2, 0, 2, 3, 0]])[:len(lengths)] + logits = torch.nn.functional.one_hot(ids, num_classes=4).float() + return (logits, []), torch.tensor([4, 6])[:len(lengths)] + + core = SimpleNamespace( + frontend=frontend, encode=encode, blank_id=0, + token_list=["", "a", "b", ""], + ctc=SimpleNamespace(ctc_lo=lambda encoded: encoded), + ) + model = SimpleNamespace(model=core, device="cpu", dtype=torch.float32) + short = torch.tensor([1., 2., 3., 4.]) + long = torch.arange(8, dtype=torch.float32) * 100 + assert transcribe_batch(model, [short, long]) == [["a", "a"], ["b", "b"]] + assert calls[0][1] == [4, 8] + assert calls[0][0].shape == (2, 8) + assert torch.equal(calls[0][0][0, 4:], torch.zeros(4)) + assert frontend.normalize_audio + assert transcribe_batch(model, [short]) == [["a", "a"]] + assert torch.equal(calls[0][0][0, :4], calls[1][0][0]) + + def fail(*args): + raise RuntimeError("encoder failed") + + core.encode = fail + with pytest.raises(RuntimeError, match="encoder failed"): + transcribe_batch(model, [short]) + assert frontend.normalize_audio + + +@pytest.mark.parametrize("batch_size", [None, 1, 2]) +def test_phonetic_grouped_batches_restore_utterance_ownership(monkeypatch, batch_size): + from batchalign.backends.phonetic.xeus import PhoneticXeusBackend + from batchalign._core.proto import PreparedAudio + + def utterance(index, start, end, *words): + return SimpleNamespace( + index=index, start_ms=start, end_ms=end, + units=[SimpleNamespace(text=word, pause=word == "(.)") for word in words], + ) + + item = SimpleNamespace( + source_id="grouping", language="eng", + audio=PreparedAudio( + pcm_f32le=base64.b64encode(b"\x00" * 32000 * 4), + sample_rate=1000, channels=1, frame_count=32000, + ), + utterances=[ + utterance(10, 0, 10000, "cat", "(.)"), + utterance(20, 10000, 19000, "dog"), + utterance(30, 21000, 22000, "fish"), + utterance(40, 30000, 32000, "cat"), + ], + ) + backend = PhoneticXeusBackend( + pronunciations={"cat": "kæt", "dog": "dɒɡ", "fish": "fɪʃ"}, + batch_size=batch_size, device="cpu", + ) + backend._model = object() + calls, ticks = [], [] + + def transcribe(model, waves): + lengths = [len(wave) for wave in waves] + calls.append(lengths) + phones = {16000: list("fɪʃ"), 32000: list("kæt"), 304000: list("kætdɒɡ")} + return [phones[length] for length in lengths] + + monkeypatch.setattr("batchalign.backends.phonetic.xeus.transcribe_batch", transcribe) + output = backend.call([item], progress=lambda *tick: ticks.append(tick))[0] + if batch_size == 2: + assert calls == [[16000, 32000], [304000]] + assert ticks == [(2, 3), (3, 3)] + else: + assert calls == [[16000], [32000], [304000]] + assert ticks == [(1, 3), (2, 3), (3, 3)] + assert [result.index for result in output.utterances] == [10, 20, 30, 40] + assert [result.ipa for result in output.utterances] == [["kæt", "(.)"], ["dɒɡ"], ["fɪʃ"], ["kæt"]] diff --git a/python/batchalign/utils/__init__.py b/python/batchalign/utils/__init__.py new file mode 100644 index 00000000..cb4a8b44 --- /dev/null +++ b/python/batchalign/utils/__init__.py @@ -0,0 +1 @@ +"""Shared utilities; sequence alignment is available in :mod:`batchalign.utils.dp`.""" diff --git a/python/batchalign/backends/morphosyntax/ud/dp.py b/python/batchalign/utils/dp.py similarity index 91% rename from python/batchalign/backends/morphosyntax/ud/dp.py rename to python/batchalign/utils/dp.py index 53884665..14fdbe10 100644 --- a/python/batchalign/backends/morphosyntax/ud/dp.py +++ b/python/batchalign/utils/dp.py @@ -1,17 +1,30 @@ -""" -dp.py -Dynamic Programming Utilities +"""Shared minimum-edit sequence alignment for Python backends. + +``align(payload, reference, tqdm=False)`` returns ordered ``Match`` and +``Extra`` records. Pass string sequences, or attach caller-owned metadata:: + + from batchalign.utils.dp import PayloadTarget, ReferenceTarget, align -Generally used for minimum-edit sequence alignment across the program. -This module now uses a Hirschberg-style divide-and-conquer aligner to -produce the same outputs while using linear space and less Python-level -overhead than the previous full-matrix implementation. + edits = align( + [PayloadTarget("cat", {"start_ms": 100})], + [ReferenceTarget("cat", 0)], + tqdm=False, + ) + # edits[0].payload == {"start_ms": 100} + # edits[0].reference_payload == 0 + +Matching costs 0, insertion/deletion 1, and substitution 2. Substitutions +produce paired extras; input metadata is returned unchanged. The normal path +calls ``batchalign._core.dp_align`` (Rust Hirschberg). A Python fallback +supports missing bindings and custom matching callbacks. """ from dataclasses import dataclass from enum import Enum from typing import Any, Callable, Iterable, List, Optional, Sequence +__all__ = ["align", "PayloadTarget", "ReferenceTarget", "Match", "Extra", "ExtraType"] + # the target to align against the reference # carries a payload which will be stitched to maching # values in the reference diff --git a/python/pyproject.toml b/python/pyproject.toml index 9dc811e3..78d82db3 100644 --- a/python/pyproject.toml +++ b/python/pyproject.toml @@ -89,6 +89,21 @@ cantonese = ["pycantonese>=4.2,<5", "tencentcloud-sdk-python-asr", "tencentcl translate = ["googletrans"] qwen3 = ["qwen-asr", "transformers>=4.57,<5", "torch>=2.0"] nllb = ["transformers>=4.57,<5", "torch>=2.0", "sentencepiece"] +# Piper Plus requires Python 3.11+; other Batchalign tasks still support 3.10. +phonetic = [ + "transformers>=4.57,<5", + "torch>=2.0,<2.13", + "torchaudio>=2.0,<2.13", + "piper-plus-g2p[all]==0.2.0; python_version >= '3.11'", + "python-mecab-ko>=1.3,<2; python_version >= '3.11' and sys_platform != 'win32'", + "eunjeon==0.4.0; python_version >= '3.11' and sys_platform == 'win32'", + "epitran>=1.35,<2", + "langcodes>=3.5,<4", + "pyyaml", + "typeguard", + "huggingface-hub", + "safetensors", +] ai = ["dspy-ai"] # HTTP daemon (`batchalign daemon`, the PyApp-bundled sidecar). The # desktop GUI's Tauri shell spawns this binary as a sidecar. `[standard]` @@ -108,6 +123,15 @@ all = [ "qwen-asr", "wtpsplit", "sentencepiece", + "piper-plus-g2p[all]==0.2.0; python_version >= '3.11'", + "python-mecab-ko>=1.3,<2; python_version >= '3.11' and sys_platform != 'win32'", + "eunjeon==0.4.0; python_version >= '3.11' and sys_platform == 'win32'", + "epitran>=1.35,<2", + "langcodes>=3.5,<4", + "pyyaml", + "typeguard", + "huggingface-hub", + "safetensors", "dspy-ai", "fastapi>=0.110", "uvicorn[standard]>=0.27", "sse-starlette>=2.0", "python-multipart>=0.0.9", ] diff --git a/python/requirements.lock.txt b/python/requirements.lock.txt index 4926d676..2f311e1b 100644 --- a/python/requirements.lock.txt +++ b/python/requirements.lock.txt @@ -914,6 +914,9 @@ diskcache==5.6.3 \ --hash=sha256:2c3a3fa2743d8535d832ec61c2054a1641f41775aa7c556758a109941e33e4fc \ --hash=sha256:5e31b2d5fbad117cc363ebaf6b689474db18a1f6438bc82358b024abd4c2ca19 # via dspy +distance==0.1.3 \ + --hash=sha256:60807584f5b6003f5c521aa73f39f51f631de3be5cccc5a1d67166fcbf0d4551 + # via g2p-en distro==1.9.0 \ --hash=sha256:2fa77c6fd8940f116ee1d6b94a2f90b13b5ea8d019b98bc8bafdcabcdd9bdbed \ --hash=sha256:7bffd925d65168f85027d8da9af6bddab658135b840670a223589bc0c8ef02b2 @@ -1091,7 +1094,9 @@ editdistance==0.8.1 \ --hash=sha256:fad081f5f86a175c1a09a4e9e45b95c9349e454c21e181e842e01c85f1f536fc \ --hash=sha256:fba945eaa0436cf40bc53d7e299dc537c7c71353379a095b7459ff4af910da33 \ --hash=sha256:fd64b58f5a7b59afd9d75982aaeeacd2a98498bf472fa0360c122ffe6ea4c871 - # via funasr + # via + # funasr + # panphon einops==0.8.2 \ --hash=sha256:54058201ac7087911181bfec4af6091bb59380360f069276601256a76af08193 \ --hash=sha256:609da665570e5e265e27283aab09e7f279ade90c4f01bcfca111f3d3e13f2827 @@ -1100,6 +1105,14 @@ emoji==2.15.0 \ --hash=sha256:205296793d66a89d88af4688fa57fd6496732eb48917a87175a023c8138995eb \ --hash=sha256:eae4ab7d86456a70a00a985125a03263a5eac54cd55e51d7e184b1ed3b6757e4 # via stanza +epitran==1.35.2 \ + --hash=sha256:14dfa6265ac51ca8972d00704712a01dcb10275e122e407da23dcec74669f0e1 \ + --hash=sha256:b5b52c35602a4982009a68b3b9bb850c61e72636e8a3c7ba5a6f9438e060fe8a + # via batchalign (python/pyproject.toml) +eunjeon==0.4.0 ; sys_platform == 'win32' \ + --hash=sha256:60865fbe28537e820ab864cd135467ea277c8182908e2bc364e83e5fd29ef07f \ + --hash=sha256:ba87fdb93ddac1a34bbb1327fc87db64ce86106abda21e3230d8e6aeb1ece44a + # via batchalign (python/pyproject.toml) execnet==2.1.2 \ --hash=sha256:63d83bfdd9a23e35b9c6a3261412324f964c2ec8dcd8d3c6916ee9373e0befcd \ --hash=sha256:67fba928dd5a544b783f6056f449e5e3931a5c378b128bc18501f7ea79e296ec @@ -1405,6 +1418,13 @@ funasr==1.3.1 \ --hash=sha256:ed813c0ecade7d24393943a82a91f7cee822178ab4bfd303adb8ad318605e0a5 \ --hash=sha256:f63050d7d625f287ec741b84a0325365699c49550a4bfd3ace4acb43e41a87a8 # via batchalign (python/pyproject.toml) +g2p-en==2.1.0 \ + --hash=sha256:2a7aabf1fc7f270fcc3349881407988c9245173c2413debbe5432f4a4f31319f \ + --hash=sha256:32ecb119827a3b10ea8c1197276f4ea4f44070ae56cbbd01f0f261875f556a58 + # via piper-plus-g2p +g2pk2==0.0.3 \ + --hash=sha256:e708316125248c10432cbead85cf52dd9ed252c423cce531f9df617386978cf2 + # via piper-plus-g2p genson==1.3.0 \ --hash=sha256:468feccd00274cc7e4c09e84b08704270ba8d95232aa280f65b986139cec67f7 \ --hash=sha256:e02db9ac2e3fd29e65b5286f7135762e2cd8a986537c075b06fc5f1517308e37 @@ -1640,6 +1660,7 @@ huggingface-hub==0.36.2 \ --hash=sha256:1934304d2fb224f8afa3b87007d58501acfda9215b334eed53072dd5e815ff7a \ --hash=sha256:48f0c8eac16145dfce371e9d2d7772854a4f591bcb56c9cf548accf531d54270 # via + # batchalign (python/pyproject.toml) # accelerate # gradio # gradio-client @@ -1675,7 +1696,9 @@ importlib-metadata==8.9.0 \ inflect==7.5.0 \ --hash=sha256:2aea70e5e70c35d8350b8097396ec155ffd68def678c7ff97f51aa69c1d92344 \ --hash=sha256:faf19801c3742ed5a05a8ce388e0d8fe1a07f8d095c82201eb904f5d27ad571f - # via datamodel-code-generator + # via + # datamodel-code-generator + # g2p-en iniconfig==2.3.0 \ --hash=sha256:c76315c77db068650d49c5b56314774a7804df16fee4402c1f19d6d15d8c4730 \ --hash=sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12 @@ -1702,7 +1725,10 @@ jaconv==0.5.0 \ jamo==0.4.1 \ --hash=sha256:d4b94fd23324c606ed2fbc4037c603e2c3a7ae9390c05d3473aea1ccb6b1c3fb \ --hash=sha256:ea65cf9d35338d0e0af48d75ff426d8a369b0ebde6f07051c3ac37256f56d025 - # via funasr + # via + # epitran + # funasr + # g2pk2 jieba==0.42.1 \ --hash=sha256:055ca12f62674fafed09427f176506079bc135638a14e23e25be909131928db2 # via funasr @@ -1977,6 +2003,10 @@ kiwisolver==1.5.0 \ --hash=sha256:fd40bb9cd0891c4c3cb1ddf83f8bbfa15731a248fdc8162669405451e2724b09 \ --hash=sha256:ff710414307fefa903e0d9bdf300972f892c23477829f49504e59834f4195398 # via matplotlib +langcodes==3.5.1 \ + --hash=sha256:40bff315e01b01d11c2ae3928dd4f5cbd74dd38f9bd912c12b9a3606c143f731 \ + --hash=sha256:b6a9c25c603804e2d169165091d0cdb23934610524a21d226e4f463e8e958a72 + # via batchalign (python/pyproject.toml) lazy-loader==0.5 \ --hash=sha256:717f9179a0dbed357012ddad50a5ad3d5e4d9a0b8712680d4e687f5e6e6ed9b3 \ --hash=sha256:ab0ea149e9c554d4ffeeb21105ac60bed7f3b4fd69b1d2360a4add51b170b005 @@ -2128,6 +2158,81 @@ mako==1.3.12 \ --hash=sha256:8f61569480282dbf557145ce441e4ba888be453c30989f879f0d652e39f53ea9 \ --hash=sha256:9f778e93289bd410bb35daadeb4fc66d95a746f0b75777b942088b7fd7af550a # via alembic +marisa-trie==1.4.1 \ + --hash=sha256:05dd3921622063b82d0c7fc51e79ba509d69a5ee72a6a2c24ee68782e5152c46 \ + --hash=sha256:0666b071851fff8b687bc6c0c899c7ce1cb6119399ed8c3c4f4526aca876a5e2 \ + --hash=sha256:0b2e53f87c01b99c59cda37411a234c704a95d12f4787aeb29572fa9302f2b93 \ + --hash=sha256:0b99c8e692cec4e172a8362832d0b1149dc126591e49643dc0c128505ea7a1cd \ + --hash=sha256:1bbdad06145ee68dd8c9280318339a401d671844420add7c48eeeddd1cc61fa8 \ + --hash=sha256:1d52339b0e879f60c8d3f52affc4a4f9acf27692c0bde63d1bd0f9f59b55dc8b \ + --hash=sha256:21ff39c29d900b44876913c96d0e2c550417450340fef5c8848111796a7f9de1 \ + --hash=sha256:26514be092531fced812709407e37d1a260a3aeb81524755cd40642860e58ef5 \ + --hash=sha256:284ff4b2a63f00e175c7fe88d18c23a556c988ce705eb8e15a65e60ad7f86a98 \ + --hash=sha256:29eb718078431518d13037830c50023333721b146ca58eb78889aabfa60f4c33 \ + --hash=sha256:3a6404610eca835cf179c4407bcaa00d7acfbf3fd7aafcc1413d3adc262b554c \ + --hash=sha256:3ac478766ff9381f1bc18f39f694388f64c20cfa8cb2b308e41ece2b4ce05467 \ + --hash=sha256:41b789fca01625288260a1db113dfb958866ae0d02887610950fd8b0c9e5dcfd \ + --hash=sha256:44ce3bdbeb7c950d463e460184fc3e18702df9ef0edb826bac672fd789fb1d20 \ + --hash=sha256:4a0b5bbd424e57b482f579d9a532bf3f9a8f2c4165175d92790cb0aa8c1c1f52 \ + --hash=sha256:4bc5d9f65d4a126dc14e32656dbc57a817ac619de731c4a64653285bf3b5e2c2 \ + --hash=sha256:4d04ddd3b1e909fed542cba20cc0c2ed4534b479ed2e8809a5182417b8165e29 \ + --hash=sha256:4d51bdd22a7238ef4d681effd7c224a267ddae054b64b1cec9ce95bbcd2b6a88 \ + --hash=sha256:4d8f3b6f7e93922d1a71c67cf285ddb5f7bc551407db2e12ba76a5f5df326449 \ + --hash=sha256:50b2bbfc6612e0b5f7bd399c3097e166e5dab2b79a58e9b956ef9127b90d2d6e \ + --hash=sha256:5154262cc60f88950f6390218e2358b4894cfcb5f22d366dfe9f2f5a7baa4b54 \ + --hash=sha256:554308b2b5b034a703c64a0c146383d9aec98538a834c4d985114cbaa987a013 \ + --hash=sha256:579d1498e6b9e8f139b36601d2ea35e9239e849ff1615f3c3fc8df8ce4d3a936 \ + --hash=sha256:59375ab1e4e4cee87d318b6b3dffa91c599c89afd920ef53428235f4326ba1d6 \ + --hash=sha256:59a5c286329a5defa33c40cce1f16c9829e4128b57ecc851ac32a7d1071913d5 \ + --hash=sha256:638fb84afc3219038648ea4814e4923914790d7e4491679ef14023459e4a8148 \ + --hash=sha256:63964dedbf49ef0d17cb32d368f13ec71ca0ec026976b1cc24cb6a993d05752a \ + --hash=sha256:63cd2870f3890f2657610ed437110713e87972da0dc4d3e6303d370c9b28d215 \ + --hash=sha256:658f49e4e825b4e4257f53f455e214cd0e161ab326e569562cc8ff8f67a48506 \ + --hash=sha256:6a5d45561a5e6563a0f934899a097d69e74111181b162de4b64cceb31f1bf44b \ + --hash=sha256:6e494d5e88da58695fa9e2efc222de808ebd36b306a2c7162256d00fb06e733e \ + --hash=sha256:71ab0be7b380d65871986d61839814153f55307ff593bac22109e65e804f07d4 \ + --hash=sha256:78a7bd2650607762a2411475544d1497e2865e0853daf0e17aa9e9555f205148 \ + --hash=sha256:7bed50d39ff1391a67b9383a7f3c458a1a0cb40fe8dd16952f813fbf8939eeff \ + --hash=sha256:7f38aad7e083d2dff8916571f31cb46a761b2d424a0b40d35881b8f108646574 \ + --hash=sha256:802f1cf5d559a3111a0784f542832e365d538e0146463d9fa5fef268302c7e86 \ + --hash=sha256:84ce9b69a0516a52169d28e27ea14f015f6daf467fc8cb661eb841565d728ccf \ + --hash=sha256:863cc513cd11b847bcd48abe65f8d9efb5d1a528adeeb323424199fed9f5cdfd \ + --hash=sha256:87e65dff37d1b9edea7bc7a8e935c851ec4934f2e56071a4501ce8db97b579a4 \ + --hash=sha256:880606b64b776c0bbd85f89cbeabf294a33dd3820971618d28bcd12fe4d1406b \ + --hash=sha256:89de7c0e6afd395b773b5adeb8ab1f315186b6538a34bbfaf73b049e3b555c2a \ + --hash=sha256:8a2dd82090d733dfb10a141bf86dcdd044dcebcf4d48c72e2048fc72b4b339d1 \ + --hash=sha256:8e759fb722a16b7db6a5fcb2ffe8c2feabf4a6143b487d21388bc5c156a79e90 \ + --hash=sha256:90a7d313a1a1edb18c43c7cfdf1f7958c2fca73746d0581087ab1683d9a4d661 \ + --hash=sha256:925b51ed02300a0db5148752e13f16d3b5d216822828f9b48b78bb2b85a81e5f \ + --hash=sha256:932e97f23815c999d8d641f79c934fe1c841eab34bd01051552822e78bba919c \ + --hash=sha256:94e9671fb7aeee0ba81988156e984b90520c721bb9c00e65d7eb4757960c7f3b \ + --hash=sha256:a56d6daf4449ae5f6825a03f9fabb97e56527fb44bc4a608944a872794f662d5 \ + --hash=sha256:a8775892a5a96df359fa8853e6132b9504dfcc2ecebd27bb617cc5be6ffeb13d \ + --hash=sha256:ab28fda06ef2e488240a17d3f9947447e7f1786ad04fb29584ab4a27fde656f4 \ + --hash=sha256:b10988ddeb8a37fd85ab03c043c5dd6fcc8f63d54af770bc27cb9722292b2a8c \ + --hash=sha256:b8315d2ec3fd52a7c439d8cf3b4fe5ea67dc46c1fd66d7bc814d2c699e831e18 \ + --hash=sha256:bc306a82f0dbece8f790bd4cfa0d7ad2dad1db9fb911395b07e7ae9862501bd2 \ + --hash=sha256:c059562d5aea86bf623a2c440b8595a86c0de553ca96986e4d36f25d07570d5b \ + --hash=sha256:c0ff2ea31a3f2ee5fcabcc77db8e5f5967d8e5614daa537262fe7617941fa262 \ + --hash=sha256:c6fbfa7f7c2f59c48be942bba848f2e378c69286abf93ecbe0b068feb1c22fbc \ + --hash=sha256:c74606bd7e0066f20cf7187de44c955ba3b4ce85159a43f8cd8b0ee982ea4c4c \ + --hash=sha256:cc68d4cdd7f1be60786888497f50c6fb8ad4f17bbec1d7accfc3fe69e725a329 \ + --hash=sha256:cde5209f0904209866e5c2ff5bdadc3d57bc9368ad3e26eac72a16d863e83dc0 \ + --hash=sha256:d8da4dea083209301430d80c8a33d0a5ecb6a270c904743505adceaae4fface2 \ + --hash=sha256:def32aa8edaec6d4229922dacb87e9d70d6bfb8ea994a13c9fcbfbee86e0b281 \ + --hash=sha256:e2c8bc2e6ede8f0ea697b050017bb65d43542507c71b32af6e6f1e14a613f9cc \ + --hash=sha256:e7f9603cc8a57dca847febf45349c51916c3e1340eb6ee064baabf181398dc79 \ + --hash=sha256:eaa3db575cb757f98d2754bcab5e1e0b2a884dc611964ac2659be13b58ef32e8 \ + --hash=sha256:ecd2f16e19f441efc6755cd09703126ee27a496b9c50f179877c00975c150189 \ + --hash=sha256:ef8f430292df0faa6a639bbba64a5e2ac09dfb967e8752d51f0bf9dd11f16b96 \ + --hash=sha256:efa5b4f8202c199ef7f4afe00ca4e406ea77b3940355595aabea8e9b393a22b1 \ + --hash=sha256:f5a6215df91c16ce4e2f674ee92a3352d4a0be30b0635ec60f4c2ef55cc7f0e2 \ + --hash=sha256:f7024c2cb001442fe04b9720be105bbe01ff2d7f357b70fa44d42272abf7da1f \ + --hash=sha256:faa9ec37e2393e86ca3cb2e568447730654dc3292097232cec1c646f257deac5 \ + --hash=sha256:faffeed161b22e343915afb2765a1f6d3cf49faa032a0725abbc69a3b22e0fba \ + --hash=sha256:fc9bc6de7197cdd1f32b72566cc7ac75c465d6f2191bba51d17edfae2b5ca8b0 \ + --hash=sha256:fe64ef107dfad7caeb1f3b49be4893572833f6c3b52312c33b40b80aaa3e9fb8 + # via epitran markdown-it-py[linkify]==4.2.0 \ --hash=sha256:04a21681d6fbb623de53f6f364d352309d4094dd4194040a10fd51833e418d49 \ --hash=sha256:9f7ebbcd14fe59494226453aed97c1070d83f8d24b6fc3a3bcf9a38092641c4a @@ -2549,6 +2654,10 @@ multidict==6.7.1 \ # via # aiohttp # yarl +munkres==1.1.4 \ + --hash=sha256:6b01867d4a8480d865aea2326e4b8f7c46431e9e55b4a2e32d989307d7bced2a \ + --hash=sha256:fc44bf3c3979dada4b6b633ddeeb8ffbe8388ee9409e4d4e8310c2da1792db03 + # via panphon mypy==2.0.0 \ --hash=sha256:0165968759c99ab79dc1a9f8aaec18e93a1bedcf7c13edd70e68dd3d5faf17cb \ --hash=sha256:106650bce72114f43019bf72197296f51c2cd47adfa9d073ea2976c247a404c5 \ @@ -2697,7 +2806,10 @@ networkx==3.6.1 \ nltk==3.9.4 \ --hash=sha256:ed03bc098a40481310320808b2db712d95d13ca65b27372f8a403949c8b523d0 \ --hash=sha256:f2fa301c3a12718ce4a0e9305c5675299da5ad9e26068218b69d692fda84828f - # via batchalign (python/pyproject.toml) + # via + # batchalign (python/pyproject.toml) + # g2p-en + # g2pk2 num2words==0.5.14 \ --hash=sha256:1c8e5b00142fc2966fd8d685001e36c4a9911e070d1b120e1beb721fa1edb33d \ --hash=sha256:b066ec18e56b6616a3b38086b5747daafbaa8868b226a36127e0451c0cf379c6 @@ -2774,6 +2886,7 @@ numpy==1.26.4 \ # contourpy # dspy # dynet38 + # g2p-en # gradio # kaldiio # librosa @@ -2784,8 +2897,10 @@ numpy==1.26.4 \ # openai-whisper # optuna # pandas + # panphon # pyannote-core # pyannote-metrics + # pyopenjtalk-plus # pytorch-metric-learning # pytorch-wpe # scikit-learn @@ -3164,9 +3279,14 @@ pandas==2.3.3 \ --hash=sha256:f8bfc0e12dc78f777f323f55c58649591b2cd0c43534e8355c51d3fede5f4dee # via # gradio + # panphon # pyannote-database # pyannote-metrics # wtpsplit +panphon==0.22.2 \ + --hash=sha256:383d477d587feba1ca5a82878e24fe77c1645d6fbe9e01c9e97d498d8a29ba35 \ + --hash=sha256:a4c65113430d0699054cb00df978c02712d3c80913a1ef67697f888d96f3a00a + # via epitran pathspec==1.1.1 \ --hash=sha256:17db5ecd524104a120e173814c90367a96a98d07c45b2e10c2f3919fff91bf5a \ --hash=sha256:a00ce642f577bf7f473932318056212bc4f8bfdf53128c78bbd5af0b9b20b189 @@ -3270,6 +3390,10 @@ pillow==12.2.0 \ # gradio # matplotlib # qwen-omni-utils +piper-plus-g2p[all]==0.2.0 \ + --hash=sha256:151662d920f27f6d674f073fabd386ec17f60aa5ab7f7a35193ba6c8a444599d \ + --hash=sha256:e99c9c7c81660ea6146542b1f035c41aeb16e9eb958f3a9b01daf64b97e19165 + # via batchalign (python/pyproject.toml) platformdirs==4.9.6 \ --hash=sha256:3bfa75b0ad0db84096ae777218481852c0ebc6c727b3168c1b9e0118e458cf0a \ --hash=sha256:e61adb1d5e5cb3441b4b7710bea7e4c12250ca49439228cc1021c00dcfac0917 @@ -3645,6 +3769,7 @@ pydantic==2.13.4 \ # gradio # litellm # openai + # pyopenjtalk-plus pydantic-core==2.46.4 \ --hash=sha256:00c603d540afdd6b80eb39f078f33ebd46211f02f33e34a32d9f053bba711de0 \ --hash=sha256:0186750b482eefa11d7f435892b09c5c606193ef3375bcf94aa00ae6bfb66262 \ @@ -3782,10 +3907,37 @@ pynndescent==0.6.0 \ --hash=sha256:7ffde0fb5b400741e055a9f7d377e3702e02250616834231f6c209e39aac24f5 \ --hash=sha256:dc8c74844e4c7f5cbd1e0cd6909da86fdc789e6ff4997336e344779c3d5538ef # via umap-learn +pyopenjtalk-plus==0.4.1.post9 \ + --hash=sha256:10d3b80ac91f193861909f08b945b11e037965f6eadb767529b975465382282c \ + --hash=sha256:1f0de887a624b86109464d9ec819e705d7412ebeda87a6abc0f828fc48091932 \ + --hash=sha256:273a7829ebbc4ddfa94c11bdf47d0aac89b418a144afb6b9d80c745416e47578 \ + --hash=sha256:3a0c9e3d7a61e72be10c6e2dc37449037e72b2b6c7a97633eb03e393836936a2 \ + --hash=sha256:58cda3998e1c8f4d820e07757803e8cd9ad324c359af59f398c30ba38fb88865 \ + --hash=sha256:5acf59e7a0750a5a35e1a76a724f257e995a71f8c06ff945f8436b8f9b62b5b9 \ + --hash=sha256:5b0e6ddebf2cf60ef480e1a9861ff2f08e6b04e70fe3ca62f349f601af6adb5b \ + --hash=sha256:5e91f4d43740e550bc8a1fb8923fd5a490f06f5fbcac0fba9280e3b7d7b954e9 \ + --hash=sha256:7122013811e02ac7c2676c4bd7e5c2d37eb6f02b11944ff1fdf3fa4667d4cd5c \ + --hash=sha256:71d06f9318d6fcdf63ba61549ea22765d674a8cfb8a29984b73915cd7cf40486 \ + --hash=sha256:75d1a93e97fe627ee88a5cc9c03af48cef440a957e66aab801e4be0bc1972384 \ + --hash=sha256:7c13d8370d82e625c3ac0712d813da498c8f145f6ff246aec482753db952a5a9 \ + --hash=sha256:856b1ee7be326015490c23fc55046688492e2d99554852e476a24a10c8e77ba0 \ + --hash=sha256:b3a991396d641cf5983bf3c49e932d6f878c8a12215cad64871c82354d22cd78 \ + --hash=sha256:c2624815bef0528628bbdc52dec298b0eb577df261f19088b7bb5018c5ca0b51 \ + --hash=sha256:cdcb0746659857554c6dad23956cad77e21f76c9f3dfa000ea2f8d4f0ba11d99 \ + --hash=sha256:d669b461203e453698da8f3a1772d15802e64cc4f0d0278d65223d30339d0302 \ + --hash=sha256:f35a6b1f977347c8b6b233728bc971fa6561cbcb1a23f6c94a79a8c763d64401 \ + --hash=sha256:f3699de7491d958685ffd2656cb0f3ac1d5c1ca469f3da7caf2e5aa4617ef3b1 \ + --hash=sha256:f7c1048013e8ce100f4f9d489d2acfc603756fb1a62ca0a2310475bc7e4fa365 \ + --hash=sha256:f98993e1e943d4313e4113992b137714c0ac8f6a383d1fd89747a592222bdddf + # via piper-plus-g2p pyparsing==3.3.2 \ --hash=sha256:850ba148bd908d7e2411587e247a1e4f0327839c40e2e5e6d05a007ecc69911d \ --hash=sha256:c777f4d763f140633dcb6d8a3eda953bf7a214dc4eff598413c070bcdc117cbc # via matplotlib +pypinyin==0.55.0 \ + --hash=sha256:b5711b3a0c6f76e67408ec6b2e3c4987a3a806b7c528076e7c7b86fcf0eaa66b \ + --hash=sha256:d53b1e8ad2cdb815fb2cb604ed3123372f5a28c6f447571244aca36fc62a286f + # via piper-plus-g2p pyproject-hooks==1.2.0 \ --hash=sha256:1e859bd5c40fae9448642dd871adf459e5e2084186e8d2c2a79a824c970da1f8 \ --hash=sha256:9e5c6bfa8dcc30091c74b0cf803c81fdd29d94f01992a7707bc97babb1141913 @@ -3812,6 +3964,58 @@ python-dotenv==1.2.2 \ # via # litellm # uvicorn +python-mecab-ko==1.3.7 ; sys_platform != 'win32' \ + --hash=sha256:010ca2297e63d08a772466dd401d36ed9914502b8794c08948427a4083b3202c \ + --hash=sha256:0933d3fcb84f6ed36cce49f1939604ac0fcaf4460441e832cb98ca1bdce74a37 \ + --hash=sha256:10ea3c549eac11cdf9e994ce65fb34653a142d04eaa519c2ba3a99646cb21991 \ + --hash=sha256:12c4b86041350024355d51dd16cb989fd027e142c8083d3b12d21b9262522054 \ + --hash=sha256:13126509630e47fc89a8c575f5af3eed1bc09370e978b331caf32325e6b98383 \ + --hash=sha256:14b070b886d864964710c6a396556d8509be2dce1618f401192fd7c213eb4608 \ + --hash=sha256:198c0b9a832966927ceceda599b8d2f38426d11d25defa0d4ed819e3d00bfa91 \ + --hash=sha256:27a03ae50aabc7f057c26ad5e4c6c4d431cf696778e45025e208d2f6b7bf115d \ + --hash=sha256:288ff89e4d1318923acecccfbb0b9d4937a8f93ac27e4868e08c778629d0522a \ + --hash=sha256:2a84df563961a6507e170f78b010716a69874fc4b00ce503280f5eb7d62ccd1c \ + --hash=sha256:3145c53772e842a046fdbf0659f0e5235e16d51b0bb8c0d3e8e078dc57d22373 \ + --hash=sha256:3387906e66109989603b877899d1ae3a0132795c9c73ad91a5e7c4c077177351 \ + --hash=sha256:3c0e98a7d94278f4f5d93f03e35cc8044460c0076ab4698b764d5c44bd897dbe \ + --hash=sha256:4321180be1e5446bb97e8f803079deb72500af7bbb7d0e2c49ec9995ec3674f5 \ + --hash=sha256:4760efe6327b5707f55db2b4a6f8fb047fe8e068577a9a913304bb0d12e7de44 \ + --hash=sha256:4fdae16e907470cec155721cc0f849a9d52e01eae316aae53101fa236069505b \ + --hash=sha256:523153de14262c413838852742541d48ad99d41ab8f6c5413a226319ee4c25ef \ + --hash=sha256:5ad754804a5a5b64b62d77a962d33ef6e931765cede89f880e02e3d18971a5bd \ + --hash=sha256:64346e4a627ad3b56647f2d6909ba52bd25b5b29f8d320944ed9dce602ba0b75 \ + --hash=sha256:644207821de8c76ff2442d84c8902dd16b239fdc80c79d0774f8b9ea446c4218 \ + --hash=sha256:661da586a6783cd60dc93ebb4dcc182e5cb3d37b98d25fe741c8eb2aabd59b30 \ + --hash=sha256:682875cd1cafeeb2946b856b1b479144b4e8d28363b6bff3ae1c8b294994742b \ + --hash=sha256:691bed2317e4cbbf4f00fc11a59d6d95412b72b9bd6eea037880df95fcd7e6a0 \ + --hash=sha256:69cbb2ac559a3169c22b1a3aa5d3c247d2f7902d9fe7dc9966189a9c7694af0b \ + --hash=sha256:7721f69381dac572a1598e5906cc5faba233ed48bc6ff8672082a519d7db0ba1 \ + --hash=sha256:782bf38e817ad54ca16dccd2e4edf083829e259aac1da3187ccc1fd305dfb503 \ + --hash=sha256:8015778e03186f8d2e7b0f1c0c9b753617d848cea2c4eba09e59e081080da92a \ + --hash=sha256:8e90e8c1009f8f6aa0dfc43c916ff481dc79aa5a7e528a41a193add9c61ac6d1 \ + --hash=sha256:99f02fb9816dda3258726b33423f0b48429582d4386529c08caa01c0d4e8365b \ + --hash=sha256:9f5e40101426b87c99ecb1268f56402f9c44f9d06271b28ccc1ec1bc6bc582ac \ + --hash=sha256:a205ca4da908df39d6d70f968426d0e9dc79274a6d34b13a5588ab52f0e12be8 \ + --hash=sha256:a456e40817dc73f58d7f11ff01af4394cdd1ceab2e98feddde625587603d65f7 \ + --hash=sha256:b8c297e6e5a8a0aacd75e9efee465d0bf7f6d1b9f0ccb9b18916e9203ea0e349 \ + --hash=sha256:c2bad59670b280548b9060c1b511f6f088c09b977355de7192e9d0044b8f724b \ + --hash=sha256:c96c719105eae24c24882fbea821df7a26c961590d06ff932599690785d7efe5 \ + --hash=sha256:d0b50438fb570299bd7e4c30549373c171b94f6400c32b0b455b37047e5ed7ed \ + --hash=sha256:d147ce60440cd04e3e113508f1c7f04ed39bcbb7991921d9c66b060709af253e \ + --hash=sha256:d7b35116e98fb736f7c9550eb1a74cfb6aa35c39b0b43cbe7a8837bfa3cd39d4 \ + --hash=sha256:d8d2539e7ea91eb0705381f75e64c626be4eba69824a8c82fbdf2c4e48a1d389 \ + --hash=sha256:da1cc9de07e75beb2d4067c1c072ecabdb293440633fc0e32f2875a14e703829 \ + --hash=sha256:e0fb84a0eda5f77dbb456fb7eba9715349668b2a9bb4235df0904620653eabda \ + --hash=sha256:e8c4347f075b8748cbc5695f6b91120b0e388344eab5d9c26d50ad3c57c35754 \ + --hash=sha256:eab31739769b1ad90fcd81f7e2319f2bc33f7b85aee3a5cec230352963678ac0 \ + --hash=sha256:eae5eb6178b06019e3773e9dde126dd29df5ed417406be5611ebdd0f8839c1e1 \ + --hash=sha256:ec22b9f8b7d5ec62d2af48d252f0172e1c4dfdf1387bad356f62b73084bac675 \ + --hash=sha256:ef5a6bb8d4611dd621436492adb140c280fe4e155097c5dcc8b1fcdd203abfb6 + # via batchalign (python/pyproject.toml) +python-mecab-ko-dic==2.1.1.post2 ; sys_platform != 'win32' \ + --hash=sha256:2c423713bdc475345ec98cd084b30759458f8f06c38a9ef94ab8687942c2cd34 \ + --hash=sha256:ef8f4e80c8976f1340a7264abb0c96f384fe059fd897584aeba0151753c6ae9b + # via python-mecab-ko python-multipart==0.0.29 \ --hash=sha256:2ddcc971cef266225f54f552d8fa10bcfbb1f14446caec199060daac59ff2d69 \ --hash=sha256:643e93849196645e2dbdd81a0f8829a23123ad7f797a84a364c6fb3563f18904 @@ -3956,6 +4160,7 @@ pyyaml==6.0.3 \ --hash=sha256:fa160448684b4e94d80416c0fa4aac48967a969efe22931448d853ada8baf926 \ --hash=sha256:fc09d0aa354569bc501d4e787133afc08552722d3ab34836a80547331bb5d4a0 # via + # batchalign (python/pyproject.toml) # accelerate # datamodel-code-generator # funasr @@ -3965,6 +4170,7 @@ pyyaml==6.0.3 \ # lightning # omegaconf # optuna + # panphon # pyannote-database # pyannote-pipeline # pytorch-lightning @@ -4101,7 +4307,9 @@ regex==2026.5.9 \ --hash=sha256:ff8d372ac2acdc048d1c19916f27ee61bc5722728458ba6ca5052f2c72d51763 # via # dspy + # epitran # nltk + # panphon # tiktoken # transformers # udtools @@ -4111,6 +4319,7 @@ requests==2.33.1 \ # via # cos-python-sdk-v5 # dspy + # epitran # funasr # google-auth # google-genai @@ -4375,6 +4584,7 @@ safetensors==0.7.0 \ --hash=sha256:e07d91d0c92a31200f25351f4acb2bc6aff7f48094e13ebb1d0fb995b54b6542 \ --hash=sha256:f4729811a6640d019a4b7ba8638ee2fd21fa5ca8c7e7bdf0fed62068fcaac737 # via + # batchalign (python/pyproject.toml) # accelerate # transformers scikit-learn==1.8.0 \ @@ -4582,7 +4792,7 @@ setuptools==81.0.0 \ # via # batchalign (python/pyproject.toml) # modelscope - # torch + # panphon shellingham==1.5.4 \ --hash=sha256:7ecfff8f2fd72616f7481040475a65b2bf8af90a56c89140852d1120324e8686 \ --hash=sha256:8dbca0739d487e5bd35ab3ca4b36e11c4078f3a234bfce294b0a0291363404de @@ -4759,6 +4969,26 @@ starlette==1.2.0 \ # fastapi # gradio # sse-starlette +sudachidict-core==20260723.1 \ + --hash=sha256:06605ab827f0ea3f37aaa5dee7fa2e5324f0a2d5bc7662e78a0a597b7eac7955 \ + --hash=sha256:2b711055dca03423869e491eca0ddbe3e17c4d7418ed738fd7c75d4e0eb9e4b1 + # via pyopenjtalk-plus +sudachipy==0.7.0 \ + --hash=sha256:0e60cefd3a7a9680206ae160bf86d266d0d60e434adac2384f8c7a6dfb0f3ba7 \ + --hash=sha256:14a2505e7d086fe9da64c0048d8e2de13cdded3aac4f1e047929cef51dba338c \ + --hash=sha256:1bf2254ee6357e0defa498630f188fefa8a543b82887077cfb1596bc3e8d4840 \ + --hash=sha256:68be139f5d7f053eba16fd3f792e4ef0c6bfed81a1b63e68b4544c7159296034 \ + --hash=sha256:6be34e856c99498e990904f7155e1a102f675cf5edfdaf7a396f26d71ce4b333 \ + --hash=sha256:8d0429db2d02d408daa7dccd062a6d8c2984a07f8e448b49a3ea694377617658 \ + --hash=sha256:907eadc1db305f9bc6e502a9acf9ff1ac871cd8d4ac3a2e0aaba3ddc60e93b2a \ + --hash=sha256:9446eba17985b1d14926afc6312fb0367687ab1b5397fa65d7e2cdd1745451c9 \ + --hash=sha256:c8306eab503d891e394fb21156e417c2f9f54cc39f56dddb8cf2b40620d470fa \ + --hash=sha256:dfc90631aa2276d2165e1995bc94b56c9f5bc27b875f43fb7d67e4a3b7346d4f \ + --hash=sha256:efd56375584dec9523fe9a26daa69da02a70ae2f2c22d25248e65a4998cae8e1 \ + --hash=sha256:f1e24f71a71816dcbc6fb759a5819cca5d603a4d49b7ade63dd0fe49468ed22c + # via + # pyopenjtalk-plus + # sudachidict-core sympy==1.13.1 ; (python_full_version >= '3.14' and platform_machine != 'x86_64') or (python_full_version >= '3.14' and sys_platform != 'darwin') \ --hash=sha256:9cebf7e04ff162015ce31c9c6c9144daa34a93bd082f54fd8f12deca4f47515f \ --hash=sha256:db36cdc64bf61b9b24578b6f7bab1ecdd2452cf008f34faa33776680c26d66f8 @@ -5180,6 +5410,7 @@ typeguard==4.4.3 \ --hash=sha256:7d8b4a3d280257fd1aa29023f22de64e29334bda0b172ff1040f05682223795e \ --hash=sha256:be72b9c85f322c20459b29060c5c099cd733d5886c4ee14297795e62b0c0d59b # via + # batchalign (python/pyproject.toml) # dspy # inflect typer==0.25.1 \ @@ -5211,6 +5442,7 @@ typing-extensions==4.15.0 \ # pyannote-core # pydantic # pydantic-core + # pyopenjtalk-plus # pytorch-lightning # referencing # sox @@ -5249,6 +5481,9 @@ umap-learn==0.5.12 \ --hash=sha256:6aff02ecac5f2aad9f3c65ee518d7ae93e1a985ae38721fdcffceee4232c33c7 \ --hash=sha256:f2a85d2a2adcb52b541bed9b27a23ca169b56bb1b23283abeebfb8dfb8a42fe5 # via funasr +unicodecsv==0.14.1 \ + --hash=sha256:018c08037d48649a0412063ff4eda26eaa81eff1546dbffa51fa5293276ff7fc + # via panphon urllib3==2.7.0 \ --hash=sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c \ --hash=sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897 diff --git a/python/uv.lock b/python/uv.lock index a4f5884c..bdecdca9 100644 --- a/python/uv.lock +++ b/python/uv.lock @@ -509,10 +509,14 @@ all = [ { name = "aliyun-python-sdk-core" }, { name = "cos-python-sdk-v5" }, { name = "dspy-ai" }, + { name = "epitran" }, + { name = "eunjeon", marker = "python_full_version >= '3.11' and sys_platform == 'win32'" }, { name = "fastapi" }, { name = "funasr" }, { name = "google-genai" }, { name = "googletrans" }, + { name = "huggingface-hub" }, + { name = "langcodes" }, { name = "nltk" }, { name = "num2words" }, { name = "numba", version = "0.62.1", source = { registry = "https://pypi.org/simple" }, marker = "platform_machine == 'x86_64' and sys_platform == 'darwin'" }, @@ -520,12 +524,16 @@ all = [ { name = "onnxruntime", version = "1.26.0", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.11'" }, { name = "openai-whisper" }, { name = "opencc" }, + { name = "piper-plus-g2p", extra = ["all"], marker = "python_full_version >= '3.11'" }, { name = "pyannote-audio" }, { name = "pycantonese" }, { name = "pycountry" }, + { name = "python-mecab-ko", marker = "python_full_version >= '3.11' and sys_platform != 'win32'" }, { name = "python-multipart" }, + { name = "pyyaml" }, { name = "qwen-asr" }, { name = "rev-ai" }, + { name = "safetensors" }, { name = "sentencepiece" }, { name = "sse-starlette" }, { name = "stanza" }, @@ -534,6 +542,7 @@ all = [ { name = "torch" }, { name = "torchaudio" }, { name = "transformers" }, + { name = "typeguard" }, { name = "uvicorn", extra = ["standard"] }, { name = "wtpsplit" }, ] @@ -579,6 +588,20 @@ nllb = [ { name = "torch" }, { name = "transformers" }, ] +phonetic = [ + { name = "epitran" }, + { name = "eunjeon", marker = "python_full_version >= '3.11' and sys_platform == 'win32'" }, + { name = "huggingface-hub" }, + { name = "langcodes" }, + { name = "piper-plus-g2p", extra = ["all"], marker = "python_full_version >= '3.11'" }, + { name = "python-mecab-ko", marker = "python_full_version >= '3.11' and sys_platform != 'win32'" }, + { name = "pyyaml" }, + { name = "safetensors" }, + { name = "torch" }, + { name = "torchaudio" }, + { name = "transformers" }, + { name = "typeguard" }, +] pyannote = [ { name = "numba", version = "0.62.1", source = { registry = "https://pypi.org/simple" }, marker = "platform_machine == 'x86_64' and sys_platform == 'darwin'" }, { name = "onnxruntime", version = "1.24.3", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, @@ -662,6 +685,10 @@ requires-dist = [ { name = "datamodel-code-generator", marker = "extra == 'dev'", specifier = ">=0.26" }, { name = "dspy-ai", marker = "extra == 'ai'" }, { name = "dspy-ai", marker = "extra == 'all'" }, + { name = "epitran", marker = "extra == 'all'", specifier = ">=1.35,<2" }, + { name = "epitran", marker = "extra == 'phonetic'", specifier = ">=1.35,<2" }, + { name = "eunjeon", marker = "python_full_version >= '3.11' and sys_platform == 'win32' and extra == 'all'", specifier = "==0.4.0" }, + { name = "eunjeon", marker = "python_full_version >= '3.11' and sys_platform == 'win32' and extra == 'phonetic'", specifier = "==0.4.0" }, { name = "fastapi", marker = "extra == 'all'", specifier = ">=0.110" }, { name = "fastapi", marker = "extra == 'api'", specifier = ">=0.110" }, { name = "funasr", marker = "extra == 'all'", specifier = "==1.3.1" }, @@ -671,6 +698,10 @@ requires-dist = [ { name = "googletrans", marker = "extra == 'all'" }, { name = "googletrans", marker = "extra == 'translate'" }, { name = "hatchling", marker = "extra == 'dev'" }, + { name = "huggingface-hub", marker = "extra == 'all'" }, + { name = "huggingface-hub", marker = "extra == 'phonetic'" }, + { name = "langcodes", marker = "extra == 'all'", specifier = ">=3.5,<4" }, + { name = "langcodes", marker = "extra == 'phonetic'", specifier = ">=3.5,<4" }, { name = "maturin", extras = ["zig"], marker = "extra == 'dev'", specifier = "==1.7.4" }, { name = "mypy", marker = "extra == 'dev'" }, { name = "nltk", marker = "extra == 'all'", specifier = ">=3.8" }, @@ -688,6 +719,8 @@ requires-dist = [ { name = "openai-whisper", marker = "extra == 'whisper'" }, { name = "opencc", marker = "extra == 'all'" }, { name = "opencc", marker = "extra == 'cantonese'" }, + { name = "piper-plus-g2p", extras = ["all"], marker = "python_full_version >= '3.11' and extra == 'all'", specifier = "==0.2.0" }, + { name = "piper-plus-g2p", extras = ["all"], marker = "python_full_version >= '3.11' and extra == 'phonetic'", specifier = "==0.2.0" }, { name = "poetry-core", marker = "extra == 'dev'" }, { name = "polars", specifier = ">=0.20" }, { name = "pyannote-audio", marker = "extra == 'all'", specifier = ">=3.4,<4" }, @@ -701,13 +734,19 @@ requires-dist = [ { name = "pydantic", specifier = ">=2" }, { name = "pytest", marker = "extra == 'dev'", specifier = ">=7" }, { name = "pytest-xdist", marker = "extra == 'dev'" }, + { name = "python-mecab-ko", marker = "python_full_version >= '3.11' and sys_platform != 'win32' and extra == 'all'", specifier = ">=1.3,<2" }, + { name = "python-mecab-ko", marker = "python_full_version >= '3.11' and sys_platform != 'win32' and extra == 'phonetic'", specifier = ">=1.3,<2" }, { name = "python-multipart", marker = "extra == 'all'", specifier = ">=0.0.9" }, { name = "python-multipart", marker = "extra == 'api'", specifier = ">=0.0.9" }, + { name = "pyyaml", marker = "extra == 'all'" }, + { name = "pyyaml", marker = "extra == 'phonetic'" }, { name = "qwen-asr", marker = "extra == 'all'" }, { name = "qwen-asr", marker = "extra == 'qwen3'" }, { name = "rev-ai", marker = "extra == 'all'" }, { name = "rev-ai", marker = "extra == 'revai'" }, { name = "rich", specifier = ">=13" }, + { name = "safetensors", marker = "extra == 'all'" }, + { name = "safetensors", marker = "extra == 'phonetic'" }, { name = "sentencepiece", marker = "extra == 'all'" }, { name = "sentencepiece", marker = "extra == 'nllb'" }, { name = "setuptools", marker = "extra == 'dev'" }, @@ -724,21 +763,26 @@ requires-dist = [ { name = "torch", marker = "extra == 'all'", specifier = ">=2.0,<2.9" }, { name = "torch", marker = "extra == 'malayalam'", specifier = ">=2.0" }, { name = "torch", marker = "extra == 'nllb'", specifier = ">=2.0" }, + { name = "torch", marker = "extra == 'phonetic'", specifier = ">=2.0,<2.13" }, { name = "torch", marker = "extra == 'pyannote'", specifier = ">=2.0,<2.9" }, { name = "torch", marker = "extra == 'qwen3'", specifier = ">=2.0" }, { name = "torch", marker = "extra == 'stanza'", specifier = ">=2.0,<2.13" }, { name = "torch", marker = "extra == 'whisper'", specifier = ">=2.0" }, { name = "torchaudio", marker = "extra == 'all'", specifier = ">=2.0,<2.9" }, { name = "torchaudio", marker = "extra == 'malayalam'", specifier = ">=2.0" }, + { name = "torchaudio", marker = "extra == 'phonetic'", specifier = ">=2.0,<2.13" }, { name = "torchaudio", marker = "extra == 'pyannote'", specifier = ">=2.0,<2.9" }, { name = "torchaudio", marker = "extra == 'whisper'", specifier = ">=2.0" }, { name = "tqdm", specifier = ">=4" }, { name = "transformers", marker = "extra == 'all'", specifier = ">=4.57,<5" }, { name = "transformers", marker = "extra == 'malayalam'", specifier = ">=4.57,<5" }, { name = "transformers", marker = "extra == 'nllb'", specifier = ">=4.57,<5" }, + { name = "transformers", marker = "extra == 'phonetic'", specifier = ">=4.57,<5" }, { name = "transformers", marker = "extra == 'qwen3'", specifier = ">=4.57,<5" }, { name = "transformers", marker = "extra == 'stanza'", specifier = ">=4.57,<5" }, { name = "transformers", marker = "extra == 'whisper'", specifier = ">=4.57,<5" }, + { name = "typeguard", marker = "extra == 'all'" }, + { name = "typeguard", marker = "extra == 'phonetic'" }, { name = "typer", specifier = ">=0.12" }, { name = "urllib3", specifier = ">=2" }, { name = "uvicorn", extras = ["standard"], marker = "extra == 'all'", specifier = ">=0.27" }, @@ -747,7 +791,7 @@ requires-dist = [ { name = "wtpsplit", marker = "extra == 'all'" }, { name = "wtpsplit", marker = "extra == 'malayalam'" }, ] -provides-extras = ["whisper", "malayalam", "stanza", "pyannote", "revai", "google", "cantonese", "translate", "qwen3", "nllb", "ai", "api", "all", "dev"] +provides-extras = ["whisper", "malayalam", "stanza", "pyannote", "revai", "google", "cantonese", "translate", "qwen3", "nllb", "phonetic", "ai", "api", "all", "dev"] [package.metadata.requires-dev] dev = [ @@ -1501,6 +1545,12 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/3f/27/4570e78fc0bf5ea0ca45eb1de3818a23787af9b390c0b0a0033a1b8236f9/diskcache-5.6.3-py3-none-any.whl", hash = "sha256:5e31b2d5fbad117cc363ebaf6b689474db18a1f6438bc82358b024abd4c2ca19", size = 45550, upload-time = "2023-08-31T06:11:58.822Z" }, ] +[[package]] +name = "distance" +version = "0.1.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/5c/1a/883e47df323437aefa0d0a92ccfb38895d9416bd0b56262c2e46a47767b8/Distance-0.1.3.tar.gz", hash = "sha256:60807584f5b6003f5c521aa73f39f51f631de3be5cccc5a1d67166fcbf0d4551", size = 180271, upload-time = "2013-11-21T00:14:34.152Z" } + [[package]] name = "distro" version = "1.9.0" @@ -1695,6 +1745,28 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e1/5e/4b5aaaabddfacfe36ba7768817bd1f71a7a810a43705e531f3ae4c690767/emoji-2.15.0-py3-none-any.whl", hash = "sha256:205296793d66a89d88af4688fa57fd6496732eb48917a87175a023c8138995eb", size = 608433, upload-time = "2025-09-21T12:13:01.197Z" }, ] +[[package]] +name = "epitran" +version = "1.35.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "jamo" }, + { name = "marisa-trie" }, + { name = "panphon" }, + { name = "regex" }, + { name = "requests" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/49/2c/22effafa0be43490d6424c550b3b641795102f5a9ecb868de0b0fef215dd/epitran-1.35.2.tar.gz", hash = "sha256:14dfa6265ac51ca8972d00704712a01dcb10275e122e407da23dcec74669f0e1", size = 165808, upload-time = "2026-06-18T20:11:47.719Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e9/a8/37e70ed73d885baedc36391d76350b3d9dbaa823b5996279fa4a384dda8f/epitran-1.35.2-py3-none-any.whl", hash = "sha256:b5b52c35602a4982009a68b3b9bb850c61e72636e8a3c7ba5a6f9438e060fe8a", size = 222057, upload-time = "2026-06-18T20:11:46.316Z" }, +] + +[[package]] +name = "eunjeon" +version = "0.4.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/68/90/3232725f974abf6d38f1e2cfd7a6b958337133b3fdc5b3e8994e03d7c2d3/eunjeon-0.4.0.tar.gz", hash = "sha256:60865fbe28537e820ab864cd135467ea277c8182908e2bc364e83e5fd29ef07f", size = 34741800, upload-time = "2019-01-17T00:42:32.268Z" } + [[package]] name = "exceptiongroup" version = "1.3.1" @@ -2052,6 +2124,34 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/57/89/61c09967f0f4f091402367215000c3c060bfeaadbbdc9aca42856d299c95/funasr-1.3.1-py3-none-any.whl", hash = "sha256:f63050d7d625f287ec741b84a0325365699c49550a4bfd3ace4acb43e41a87a8", size = 811975, upload-time = "2026-01-26T13:07:41.866Z" }, ] +[[package]] +name = "g2p-en" +version = "2.1.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "distance", marker = "python_full_version >= '3.11'" }, + { name = "inflect", marker = "python_full_version >= '3.11'" }, + { name = "nltk", marker = "python_full_version >= '3.11'" }, + { name = "numpy", version = "1.26.4", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.11' and platform_machine == 'x86_64' and sys_platform == 'darwin'" }, + { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "(python_full_version >= '3.11' and platform_machine != 'x86_64') or (python_full_version >= '3.11' and sys_platform != 'darwin')" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/5f/22/2c7acbe6164ed6cfd4301e9ad2dbde69c68d22268a0f9b5b0ee6052ed3ab/g2p_en-2.1.0.tar.gz", hash = "sha256:32ecb119827a3b10ea8c1197276f4ea4f44070ae56cbbd01f0f261875f556a58", size = 3116166, upload-time = "2019-12-31T01:16:12.753Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d7/d9/b77dc634a7a0c0c97716ba97dd0a28cbfa6267c96f359c4f27ed71cbd284/g2p_en-2.1.0-py3-none-any.whl", hash = "sha256:2a7aabf1fc7f270fcc3349881407988c9245173c2413debbe5432f4a4f31319f", size = 3117464, upload-time = "2019-12-31T01:16:03.286Z" }, +] + +[[package]] +name = "g2pk2" +version = "0.0.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "jamo", marker = "python_full_version >= '3.11'" }, + { name = "nltk", marker = "python_full_version >= '3.11'" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/53/52/6b77a9c3e4ea32dfb53238422dabf4fa977681d2bf4087bff53c084ab7c5/g2pk2-0.0.3-py3-none-any.whl", hash = "sha256:e708316125248c10432cbead85cf52dd9ed252c423cce531f9df617386978cf2", size = 25716, upload-time = "2023-08-18T17:05:33.705Z" }, +] + [[package]] name = "genson" version = "1.3.0" @@ -3020,6 +3120,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/0a/dd/8050c947d435c8d4bc94e3252f4d8bb8a76cfb424f043a8680be637a57f1/kiwisolver-1.5.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:59cd8683f575d96df5bb48f6add94afc055012c29e28124fcae2b63661b9efb1", size = 73558, upload-time = "2026-03-09T13:15:52.112Z" }, ] +[[package]] +name = "langcodes" +version = "3.5.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a9/75/f9edc5d72945019312f359e69ded9f82392a81d49c5051ed3209b100c0d2/langcodes-3.5.1.tar.gz", hash = "sha256:40bff315e01b01d11c2ae3928dd4f5cbd74dd38f9bd912c12b9a3606c143f731", size = 191084, upload-time = "2025-12-02T16:22:01.627Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/dd/c1/d10b371bcba7abce05e2b33910e39c33cfa496a53f13640b7b8e10bb4d2b/langcodes-3.5.1-py3-none-any.whl", hash = "sha256:b6a9c25c603804e2d169165091d0cdb23934610524a21d226e4f463e8e958a72", size = 183050, upload-time = "2025-12-02T16:21:59.954Z" }, +] + [[package]] name = "lazy-loader" version = "0.5" @@ -3347,6 +3456,77 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/bc/b1/a0ec7a5a9db730a08daef1fdfb8090435b82465abbf758a596f0ea88727e/mako-1.3.12-py3-none-any.whl", hash = "sha256:8f61569480282dbf557145ce441e4ba888be453c30989f879f0d652e39f53ea9", size = 78521, upload-time = "2026-04-28T19:01:10.393Z" }, ] +[[package]] +name = "marisa-trie" +version = "1.4.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/77/5d/e235921b5b74818cb65b557fa05cc6201c2c1612d4866ff75c835bcf808d/marisa_trie-1.4.1.tar.gz", hash = "sha256:44ce3bdbeb7c950d463e460184fc3e18702df9ef0edb826bac672fd789fb1d20", size = 261581, upload-time = "2026-04-08T07:17:52.991Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ed/56/ec51c02b083ccd25ce3f1cf13b9c575c05497d18f9386ac51314bc62fea8/marisa_trie-1.4.1-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:bc306a82f0dbece8f790bd4cfa0d7ad2dad1db9fb911395b07e7ae9862501bd2", size = 209920, upload-time = "2026-04-08T07:16:12.757Z" }, + { url = "https://files.pythonhosted.org/packages/87/96/3fc7b2e94c1da93636582acc7d56eb186c4d2cb2c01b0b2c2a177ca11061/marisa_trie-1.4.1-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:def32aa8edaec6d4229922dacb87e9d70d6bfb8ea994a13c9fcbfbee86e0b281", size = 193667, upload-time = "2026-04-08T07:16:14.111Z" }, + { url = "https://files.pythonhosted.org/packages/43/df/309e88b3c2bbf2abdc9f62a4ebc313b24130edddecca2889db9dbd74c765/marisa_trie-1.4.1-cp310-cp310-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:89de7c0e6afd395b773b5adeb8ab1f315186b6538a34bbfaf73b049e3b555c2a", size = 1457943, upload-time = "2026-04-08T07:16:15.587Z" }, + { url = "https://files.pythonhosted.org/packages/c1/e9/1197d04f35791607d01c010ae0de55ca53de574f21774625af0da3b573f8/marisa_trie-1.4.1-cp310-cp310-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:ecd2f16e19f441efc6755cd09703126ee27a496b9c50f179877c00975c150189", size = 1480962, upload-time = "2026-04-08T07:16:17.342Z" }, + { url = "https://files.pythonhosted.org/packages/e8/ea/ff378bf751f053cb4dd6082cc6bcffef62af765aa39f974250323586015a/marisa_trie-1.4.1-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:ef8f430292df0faa6a639bbba64a5e2ac09dfb967e8752d51f0bf9dd11f16b96", size = 2387724, upload-time = "2026-04-08T07:16:19.079Z" }, + { url = "https://files.pythonhosted.org/packages/c3/cc/726372806ddd3dd4f7458b15ca890eb52f4a33e86260cff868ddc4a0f797/marisa_trie-1.4.1-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:c0ff2ea31a3f2ee5fcabcc77db8e5f5967d8e5614daa537262fe7617941fa262", size = 2489031, upload-time = "2026-04-08T07:16:20.754Z" }, + { url = "https://files.pythonhosted.org/packages/12/4c/64698d167825e53f1cd375db6d9357a87ecec1a00d12f6f3651f8a858457/marisa_trie-1.4.1-cp310-cp310-win32.whl", hash = "sha256:b8315d2ec3fd52a7c439d8cf3b4fe5ea67dc46c1fd66d7bc814d2c699e831e18", size = 141020, upload-time = "2026-04-08T07:16:22.156Z" }, + { url = "https://files.pythonhosted.org/packages/fa/81/bdd4ce80ab15bc422377f123ac6c49b91396a71623cac7bc8198f1c4e016/marisa_trie-1.4.1-cp310-cp310-win_amd64.whl", hash = "sha256:1bbdad06145ee68dd8c9280318339a401d671844420add7c48eeeddd1cc61fa8", size = 174452, upload-time = "2026-04-08T07:16:23.438Z" }, + { url = "https://files.pythonhosted.org/packages/1e/87/c65aeaeed6d8563a7522215bdc9d676a6978f8bb27071d7b59a5337a2ff9/marisa_trie-1.4.1-cp310-cp310-win_arm64.whl", hash = "sha256:b10988ddeb8a37fd85ab03c043c5dd6fcc8f63d54af770bc27cb9722292b2a8c", size = 142249, upload-time = "2026-04-08T07:16:24.578Z" }, + { url = "https://files.pythonhosted.org/packages/3d/94/1ad851729ba0cdbc269bdd72b4b725cfbf5a25e4186fd12e56836d2d52ba/marisa_trie-1.4.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:579d1498e6b9e8f139b36601d2ea35e9239e849ff1615f3c3fc8df8ce4d3a936", size = 208644, upload-time = "2026-04-08T07:16:25.926Z" }, + { url = "https://files.pythonhosted.org/packages/11/67/2a8870ef1c42412ace3d656830902893fad6239434cde749f6654f907b41/marisa_trie-1.4.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:3a6404610eca835cf179c4407bcaa00d7acfbf3fd7aafcc1413d3adc262b554c", size = 192740, upload-time = "2026-04-08T07:16:26.971Z" }, + { url = "https://files.pythonhosted.org/packages/b7/3f/d1d67e1ae5058dfcbb3756f75a26ea203acae875584636d360fd4a38b248/marisa_trie-1.4.1-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:284ff4b2a63f00e175c7fe88d18c23a556c988ce705eb8e15a65e60ad7f86a98", size = 1506335, upload-time = "2026-04-08T07:16:28.378Z" }, + { url = "https://files.pythonhosted.org/packages/da/23/086b81133baccbc05feb2fbc4113c70775352f3b9851dc8a20d69b2db44f/marisa_trie-1.4.1-cp311-cp311-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:29eb718078431518d13037830c50023333721b146ca58eb78889aabfa60f4c33", size = 1529146, upload-time = "2026-04-08T07:16:30.061Z" }, + { url = "https://files.pythonhosted.org/packages/b8/a0/c717ffb7697f059cde03b75c9facf6fd4ee18ce720513156bb03e875d215/marisa_trie-1.4.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:3ac478766ff9381f1bc18f39f694388f64c20cfa8cb2b308e41ece2b4ce05467", size = 2436516, upload-time = "2026-04-08T07:16:31.452Z" }, + { url = "https://files.pythonhosted.org/packages/fd/d7/c5f5cd9eb32a34447971ee3430410b19087564137d62da7bd348b40e684a/marisa_trie-1.4.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:f7024c2cb001442fe04b9720be105bbe01ff2d7f357b70fa44d42272abf7da1f", size = 2532453, upload-time = "2026-04-08T07:16:32.711Z" }, + { url = "https://files.pythonhosted.org/packages/32/56/84fe4788b29a81e00efb4c97f3c1b7cae0dbda14369dc7ee2659823bfd83/marisa_trie-1.4.1-cp311-cp311-win32.whl", hash = "sha256:c059562d5aea86bf623a2c440b8595a86c0de553ca96986e4d36f25d07570d5b", size = 140594, upload-time = "2026-04-08T07:16:33.883Z" }, + { url = "https://files.pythonhosted.org/packages/d5/b0/3f793c5f7727ca1d36f9d53b9fc48804942918a9f460b6d46988770677b2/marisa_trie-1.4.1-cp311-cp311-win_amd64.whl", hash = "sha256:c74606bd7e0066f20cf7187de44c955ba3b4ce85159a43f8cd8b0ee982ea4c4c", size = 175072, upload-time = "2026-04-08T07:16:35.254Z" }, + { url = "https://files.pythonhosted.org/packages/fe/8c/ac1851b7c9ce881343bf0639c902aaa2b767a31dc79d6af5950413a6ba66/marisa_trie-1.4.1-cp311-cp311-win_arm64.whl", hash = "sha256:59a5c286329a5defa33c40cce1f16c9829e4128b57ecc851ac32a7d1071913d5", size = 142560, upload-time = "2026-04-08T07:16:36.425Z" }, + { url = "https://files.pythonhosted.org/packages/b3/b7/89811f7eba6e92386279376df81cfa281ab99e30f7e4f5a5e04d8dba6b99/marisa_trie-1.4.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:63964dedbf49ef0d17cb32d368f13ec71ca0ec026976b1cc24cb6a993d05752a", size = 206731, upload-time = "2026-04-08T07:16:37.439Z" }, + { url = "https://files.pythonhosted.org/packages/f7/8b/cc34313149486dfc13e84303e12d61fd55788b37d92c3e082cc3d142e776/marisa_trie-1.4.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:87e65dff37d1b9edea7bc7a8e935c851ec4934f2e56071a4501ce8db97b579a4", size = 190988, upload-time = "2026-04-08T07:16:38.686Z" }, + { url = "https://files.pythonhosted.org/packages/15/0c/376e21c62bd0e658a5e9f6b8912f3116591778c639857ad374c7639ceebe/marisa_trie-1.4.1-cp312-cp312-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7bed50d39ff1391a67b9383a7f3c458a1a0cb40fe8dd16952f813fbf8939eeff", size = 1471836, upload-time = "2026-04-08T07:16:39.88Z" }, + { url = "https://files.pythonhosted.org/packages/bb/95/cd6e73d0857608f2946f3bc5ccac86488073b96fb37bc1b45b0184268bed/marisa_trie-1.4.1-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:4d51bdd22a7238ef4d681effd7c224a267ddae054b64b1cec9ce95bbcd2b6a88", size = 1516414, upload-time = "2026-04-08T07:16:41.4Z" }, + { url = "https://files.pythonhosted.org/packages/51/73/339e8fab2e8cea88e9e0fd78aeb8ccdd3f8656d228dae2cb698f667a0fe7/marisa_trie-1.4.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:d8da4dea083209301430d80c8a33d0a5ecb6a270c904743505adceaae4fface2", size = 2394325, upload-time = "2026-04-08T07:16:43.139Z" }, + { url = "https://files.pythonhosted.org/packages/5c/d7/0ba8bcaeee68a8e6cbc61b47825370a6c8a523ab16ea42e8728dec2213bc/marisa_trie-1.4.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:e7f9603cc8a57dca847febf45349c51916c3e1340eb6ee064baabf181398dc79", size = 2510974, upload-time = "2026-04-08T07:16:44.848Z" }, + { url = "https://files.pythonhosted.org/packages/23/ef/fa342fdbc0c055030b93007dff5393675705071a14621e3f81bfb52eb970/marisa_trie-1.4.1-cp312-cp312-win32.whl", hash = "sha256:63cd2870f3890f2657610ed437110713e87972da0dc4d3e6303d370c9b28d215", size = 138890, upload-time = "2026-04-08T07:16:46.871Z" }, + { url = "https://files.pythonhosted.org/packages/73/7d/419114325f1bb4c2202c20f19f424f9dea1cc38de7cd1fae60d991e99b69/marisa_trie-1.4.1-cp312-cp312-win_amd64.whl", hash = "sha256:fc9bc6de7197cdd1f32b72566cc7ac75c465d6f2191bba51d17edfae2b5ca8b0", size = 168513, upload-time = "2026-04-08T07:16:48.475Z" }, + { url = "https://files.pythonhosted.org/packages/8f/f8/ae0dcbf79498b7aa00dae740982c9812fa95339bc6549ea63b4ad15eeb58/marisa_trie-1.4.1-cp312-cp312-win_arm64.whl", hash = "sha256:59375ab1e4e4cee87d318b6b3dffa91c599c89afd920ef53428235f4326ba1d6", size = 139710, upload-time = "2026-04-08T07:16:49.863Z" }, + { url = "https://files.pythonhosted.org/packages/91/df/a6b189cfdfc45fc402833fa067b1625a8ec4ef5446a8d7c08c5c84ea835e/marisa_trie-1.4.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:cde5209f0904209866e5c2ff5bdadc3d57bc9368ad3e26eac72a16d863e83dc0", size = 206631, upload-time = "2026-04-08T07:16:51.18Z" }, + { url = "https://files.pythonhosted.org/packages/4f/1b/7b03330888306166e96801acb5086eaf5ddc112d2ab8c03c8de478da7346/marisa_trie-1.4.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:21ff39c29d900b44876913c96d0e2c550417450340fef5c8848111796a7f9de1", size = 190110, upload-time = "2026-04-08T07:16:52.276Z" }, + { url = "https://files.pythonhosted.org/packages/dc/41/6ae103ef7448320a7324f9866a253d595159ffe367c11e06baabb92ca4d2/marisa_trie-1.4.1-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:84ce9b69a0516a52169d28e27ea14f015f6daf467fc8cb661eb841565d728ccf", size = 1474539, upload-time = "2026-04-08T07:16:53.924Z" }, + { url = "https://files.pythonhosted.org/packages/1b/dc/cbb5e8416ff5d193847b0e838b0b525773af7b6f4e1e4a33728d1b097fcb/marisa_trie-1.4.1-cp313-cp313-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:eaa3db575cb757f98d2754bcab5e1e0b2a884dc611964ac2659be13b58ef32e8", size = 1501031, upload-time = "2026-04-08T07:16:55.554Z" }, + { url = "https://files.pythonhosted.org/packages/88/5c/ed86ad8683237dff8cdeb117b3b0664005e05bc221535301621ed474857d/marisa_trie-1.4.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:71ab0be7b380d65871986d61839814153f55307ff593bac22109e65e804f07d4", size = 2397211, upload-time = "2026-04-08T07:16:57.199Z" }, + { url = "https://files.pythonhosted.org/packages/49/a3/2596d55ee48ed15a4d0de5a9ebd27de49888a029ffa521717c282df63efc/marisa_trie-1.4.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:658f49e4e825b4e4257f53f455e214cd0e161ab326e569562cc8ff8f67a48506", size = 2498742, upload-time = "2026-04-08T07:16:58.779Z" }, + { url = "https://files.pythonhosted.org/packages/fe/d1/50c0ed09c99e4cd3ff7276f1a50a148c7bd4e585271215d3a05f74228bf9/marisa_trie-1.4.1-cp313-cp313-win32.whl", hash = "sha256:a56d6daf4449ae5f6825a03f9fabb97e56527fb44bc4a608944a872794f662d5", size = 138706, upload-time = "2026-04-08T07:17:00.258Z" }, + { url = "https://files.pythonhosted.org/packages/68/67/b35e8b14757ce5daffd5c4c1ab0bb9b3e3c7611e82fe5ab2707489a176c4/marisa_trie-1.4.1-cp313-cp313-win_amd64.whl", hash = "sha256:6a5d45561a5e6563a0f934899a097d69e74111181b162de4b64cceb31f1bf44b", size = 168903, upload-time = "2026-04-08T07:17:01.419Z" }, + { url = "https://files.pythonhosted.org/packages/da/91/bd06914afcb70710f684be44cf5435742d175d49c3de021ee62f7eb8c4e4/marisa_trie-1.4.1-cp313-cp313-win_arm64.whl", hash = "sha256:ab28fda06ef2e488240a17d3f9947447e7f1786ad04fb29584ab4a27fde656f4", size = 139620, upload-time = "2026-04-08T07:17:02.354Z" }, + { url = "https://files.pythonhosted.org/packages/75/b5/3823948064c63fd76777910b45481c6e251c9d4c3f261ca23d51b758dc91/marisa_trie-1.4.1-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:efa5b4f8202c199ef7f4afe00ca4e406ea77b3940355595aabea8e9b393a22b1", size = 213383, upload-time = "2026-04-08T07:17:03.766Z" }, + { url = "https://files.pythonhosted.org/packages/ae/f1/5c77eea2ff285e47e8a523385f023075a452455ffa55804df95e4b569a12/marisa_trie-1.4.1-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:a8775892a5a96df359fa8853e6132b9504dfcc2ecebd27bb617cc5be6ffeb13d", size = 202037, upload-time = "2026-04-08T07:17:04.745Z" }, + { url = "https://files.pythonhosted.org/packages/2a/0d/0ea5e7f0aaa11c29aadaf121b6de275a6807d9712ef28ea4a6025bcab2e4/marisa_trie-1.4.1-cp313-cp313t-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:cc68d4cdd7f1be60786888497f50c6fb8ad4f17bbec1d7accfc3fe69e725a329", size = 1542413, upload-time = "2026-04-08T07:17:06.247Z" }, + { url = "https://files.pythonhosted.org/packages/8a/a9/fe6aa360eba3178cd74796b5e0d6d07a1afabe5b09b54fa8c2aa693a53c8/marisa_trie-1.4.1-cp313-cp313t-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8e759fb722a16b7db6a5fcb2ffe8c2feabf4a6143b487d21388bc5c156a79e90", size = 1545503, upload-time = "2026-04-08T07:17:07.501Z" }, + { url = "https://files.pythonhosted.org/packages/60/42/2d80e091d2b92f7175be488f554565d0e0a432b1d61c4804177e0ec9ecce/marisa_trie-1.4.1-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:0666b071851fff8b687bc6c0c899c7ce1cb6119399ed8c3c4f4526aca876a5e2", size = 2447028, upload-time = "2026-04-08T07:17:09.29Z" }, + { url = "https://files.pythonhosted.org/packages/28/8b/f9cb7fab4a0053dd9caed573791ab292f8b890a31bd5f159ef21dbe40e63/marisa_trie-1.4.1-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:6e494d5e88da58695fa9e2efc222de808ebd36b306a2c7162256d00fb06e733e", size = 2536022, upload-time = "2026-04-08T07:17:11.222Z" }, + { url = "https://files.pythonhosted.org/packages/64/00/ad53cf35464937b7719ed7a6e21354418f998df9002944cfa5f14903f460/marisa_trie-1.4.1-cp313-cp313t-win32.whl", hash = "sha256:50b2bbfc6612e0b5f7bd399c3097e166e5dab2b79a58e9b956ef9127b90d2d6e", size = 153959, upload-time = "2026-04-08T07:17:12.799Z" }, + { url = "https://files.pythonhosted.org/packages/02/e5/89ae70c984c178a5cf0fa95c8c47aab129b73641ec3729ff56cc5039abc1/marisa_trie-1.4.1-cp313-cp313t-win_amd64.whl", hash = "sha256:932e97f23815c999d8d641f79c934fe1c841eab34bd01051552822e78bba919c", size = 186646, upload-time = "2026-04-08T07:17:13.88Z" }, + { url = "https://files.pythonhosted.org/packages/82/84/514da5bd3ee051e800caf1ce687b7727a92b643416cc678dc47d3eface97/marisa_trie-1.4.1-cp313-cp313t-win_arm64.whl", hash = "sha256:0b2e53f87c01b99c59cda37411a234c704a95d12f4787aeb29572fa9302f2b93", size = 146566, upload-time = "2026-04-08T07:17:15.175Z" }, + { url = "https://files.pythonhosted.org/packages/af/94/83532d3ddb47db51571f6005bf227676abb9f0b490e56faefeaa74303ad1/marisa_trie-1.4.1-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:4d8f3b6f7e93922d1a71c67cf285ddb5f7bc551407db2e12ba76a5f5df326449", size = 207601, upload-time = "2026-04-08T07:17:16.582Z" }, + { url = "https://files.pythonhosted.org/packages/52/0b/2a670dd3c163836e516181469e34d9abdd9fee7e88183bfd66dd3cdc0ecc/marisa_trie-1.4.1-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:638fb84afc3219038648ea4814e4923914790d7e4491679ef14023459e4a8148", size = 192341, upload-time = "2026-04-08T07:17:17.773Z" }, + { url = "https://files.pythonhosted.org/packages/33/60/fddeebbc819873e3f23a09d01226e12911ee5d23252473bdf038aaf1c928/marisa_trie-1.4.1-cp314-cp314-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f5a6215df91c16ce4e2f674ee92a3352d4a0be30b0635ec60f4c2ef55cc7f0e2", size = 1474315, upload-time = "2026-04-08T07:17:19.117Z" }, + { url = "https://files.pythonhosted.org/packages/66/61/a4f1e474809cbb08df715d63f7680568d457c2b543a53964b969d94be7b0/marisa_trie-1.4.1-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0b99c8e692cec4e172a8362832d0b1149dc126591e49643dc0c128505ea7a1cd", size = 1491206, upload-time = "2026-04-08T07:17:20.382Z" }, + { url = "https://files.pythonhosted.org/packages/51/34/1fbe625a8644ee49a6829dc2a2a22d6be6ad2d26b8b06ffe22f8aa50b6eb/marisa_trie-1.4.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:880606b64b776c0bbd85f89cbeabf294a33dd3820971618d28bcd12fe4d1406b", size = 2401513, upload-time = "2026-04-08T07:17:21.769Z" }, + { url = "https://files.pythonhosted.org/packages/72/ec/edf909d5770c75aef4c94bae7dbaba56e9877b822ef6fa82b07958dd45bb/marisa_trie-1.4.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:554308b2b5b034a703c64a0c146383d9aec98538a834c4d985114cbaa987a013", size = 2489796, upload-time = "2026-04-08T07:17:23.477Z" }, + { url = "https://files.pythonhosted.org/packages/be/ea/e79d0471e32a0681d6f9f59560f5dd9748202e1b415340cf8bba338fe123/marisa_trie-1.4.1-cp314-cp314-win32.whl", hash = "sha256:c6fbfa7f7c2f59c48be942bba848f2e378c69286abf93ecbe0b068feb1c22fbc", size = 142994, upload-time = "2026-04-08T07:17:24.784Z" }, + { url = "https://files.pythonhosted.org/packages/13/7f/acb58830aed8dac2509a06ef476f6e59767eca09f08e23f3c8eaf7cfc323/marisa_trie-1.4.1-cp314-cp314-win_amd64.whl", hash = "sha256:e2c8bc2e6ede8f0ea697b050017bb65d43542507c71b32af6e6f1e14a613f9cc", size = 173279, upload-time = "2026-04-08T07:17:26.159Z" }, + { url = "https://files.pythonhosted.org/packages/3d/ac/6a9415c619b3b6fc3bc5a2cc93b3e5e8d61804c95604f20379468ee2214a/marisa_trie-1.4.1-cp314-cp314-win_arm64.whl", hash = "sha256:4bc5d9f65d4a126dc14e32656dbc57a817ac619de731c4a64653285bf3b5e2c2", size = 146116, upload-time = "2026-04-08T07:17:27.225Z" }, + { url = "https://files.pythonhosted.org/packages/30/f1/3e9f3be8ccfd97ddde488a4eb9c7b7a603f1f7ee73d13814483146238c16/marisa_trie-1.4.1-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:faffeed161b22e343915afb2765a1f6d3cf49faa032a0725abbc69a3b22e0fba", size = 214797, upload-time = "2026-04-08T07:17:28.191Z" }, + { url = "https://files.pythonhosted.org/packages/6c/c3/a11da7513a7eb99e9dcbe46eee97e53c9cb66f4ed253caed46f5d2492f0a/marisa_trie-1.4.1-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:1d52339b0e879f60c8d3f52affc4a4f9acf27692c0bde63d1bd0f9f59b55dc8b", size = 203005, upload-time = "2026-04-08T07:17:29.28Z" }, + { url = "https://files.pythonhosted.org/packages/54/a6/ec11c67e5914873aa5e464c0a83f8b31b3bc6961a71d72a8cafe3591dbcb/marisa_trie-1.4.1-cp314-cp314t-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:05dd3921622063b82d0c7fc51e79ba509d69a5ee72a6a2c24ee68782e5152c46", size = 1543416, upload-time = "2026-04-08T07:17:30.381Z" }, + { url = "https://files.pythonhosted.org/packages/b3/57/14c5a667663797add840a755bef63b82eca1a07e64fc046875e294af5af0/marisa_trie-1.4.1-cp314-cp314t-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:faa9ec37e2393e86ca3cb2e568447730654dc3292097232cec1c646f257deac5", size = 1544901, upload-time = "2026-04-08T07:17:31.776Z" }, + { url = "https://files.pythonhosted.org/packages/05/d6/b4d74f2575df3be12cbbab084dbf1db70ca6c9934ca3946d79fc45a8e83b/marisa_trie-1.4.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:fe64ef107dfad7caeb1f3b49be4893572833f6c3b52312c33b40b80aaa3e9fb8", size = 2447903, upload-time = "2026-04-08T07:17:33.497Z" }, + { url = "https://files.pythonhosted.org/packages/ec/27/e923abac9c578dd6b8726e03c433af6847538fb4db252befd68b9f5a0550/marisa_trie-1.4.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:7f38aad7e083d2dff8916571f31cb46a761b2d424a0b40d35881b8f108646574", size = 2537191, upload-time = "2026-04-08T07:17:35.242Z" }, + { url = "https://files.pythonhosted.org/packages/53/9d/b553a2a3b1809f376a5ecf03b1e3f267e3c95c39fd0448d275b1b5399452/marisa_trie-1.4.1-cp314-cp314t-win32.whl", hash = "sha256:5154262cc60f88950f6390218e2358b4894cfcb5f22d366dfe9f2f5a7baa4b54", size = 160728, upload-time = "2026-04-08T07:17:36.576Z" }, + { url = "https://files.pythonhosted.org/packages/61/32/dcff70d08a146ca94a2934ca9e3ff641bf1f0e5a7785714874ac5f085c57/marisa_trie-1.4.1-cp314-cp314t-win_amd64.whl", hash = "sha256:4d04ddd3b1e909fed542cba20cc0c2ed4534b479ed2e8809a5182417b8165e29", size = 199138, upload-time = "2026-04-08T07:17:37.69Z" }, + { url = "https://files.pythonhosted.org/packages/ba/ab/693f98813cc0484c99da4fe9b73ca65b7dd15ab8eb4f23d9a6a77c8852f9/marisa_trie-1.4.1-cp314-cp314t-win_arm64.whl", hash = "sha256:41b789fca01625288260a1db113dfb958866ae0d02887610950fd8b0c9e5dcfd", size = 151453, upload-time = "2026-04-08T07:17:39.065Z" }, +] + [[package]] name = "markdown-it-py" version = "4.2.0" @@ -3820,6 +4000,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/81/08/7036c080d7117f28a4af526d794aab6a84463126db031b007717c1a6676e/multidict-6.7.1-py3-none-any.whl", hash = "sha256:55d97cc6dae627efa6a6e548885712d4864b81110ac76fa4e534c03819fa4a56", size = 12319, upload-time = "2026-01-26T02:46:44.004Z" }, ] +[[package]] +name = "munkres" +version = "1.1.4" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/fd/41/6a3d0ef908f47d07c31e5d1c2504388c27c39b10b8cf610175b5a789a5c1/munkres-1.1.4.tar.gz", hash = "sha256:fc44bf3c3979dada4b6b633ddeeb8ffbe8388ee9409e4d4e8310c2da1792db03", size = 14047, upload-time = "2020-09-15T15:12:20.956Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/90/ab/0301c945a704218bc9435f0e3c88884f6b19ef234d8899fb47ce1ccfd0c9/munkres-1.1.4-py2.py3-none-any.whl", hash = "sha256:6b01867d4a8480d865aea2326e4b8f7c46431e9e55b4a2e32d989307d7bced2a", size = 7015, upload-time = "2020-09-15T15:12:19.627Z" }, +] + [[package]] name = "mypy" version = "2.1.0" @@ -4906,6 +5095,28 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/0f/54/68a0978d1ef8502b8492099beaa6e7a0c1b32e3b5d4f677f5810cb08711c/pandas-3.0.3-cp314-cp314t-win_arm64.whl", hash = "sha256:b2c95f8bfc1ee412bf482605d7bfd30c12d1d26bd59fdd91efeef1d4718decb1", size = 9466464, upload-time = "2026-05-11T18:54:22.754Z" }, ] +[[package]] +name = "panphon" +version = "0.22.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "editdistance" }, + { name = "munkres" }, + { name = "numpy", version = "1.26.4", source = { registry = "https://pypi.org/simple" }, marker = "platform_machine == 'x86_64' and sys_platform == 'darwin'" }, + { name = "numpy", version = "2.2.6", source = { registry = "https://pypi.org/simple" }, marker = "(python_full_version < '3.11' and platform_machine != 'x86_64') or (python_full_version < '3.11' and sys_platform != 'darwin')" }, + { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "(python_full_version >= '3.11' and platform_machine != 'x86_64') or (python_full_version >= '3.11' and sys_platform != 'darwin')" }, + { name = "pandas", version = "2.3.3", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11' or (python_full_version >= '3.14' and platform_machine == 'x86_64' and sys_platform == 'darwin')" }, + { name = "pandas", version = "3.0.3", source = { registry = "https://pypi.org/simple" }, marker = "(python_full_version >= '3.11' and python_full_version < '3.14') or (python_full_version >= '3.11' and platform_machine != 'x86_64') or (python_full_version >= '3.11' and sys_platform != 'darwin')" }, + { name = "pyyaml" }, + { name = "regex" }, + { name = "setuptools" }, + { name = "unicodecsv" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/fe/00/846bf96c41a5dcc2ead957ac7229931c456091735c73d98454ad3c109931/panphon-0.22.2.tar.gz", hash = "sha256:383d477d587feba1ca5a82878e24fe77c1645d6fbe9e01c9e97d498d8a29ba35", size = 79244, upload-time = "2025-06-12T22:23:05.173Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/72/2a/05a47a6c291414f5a66901f8d1f6ba68e8ec9fccb341d56fa0b31479d41d/panphon-0.22.2-py2.py3-none-any.whl", hash = "sha256:a4c65113430d0699054cb00df978c02712d3c80913a1ef67697f888d96f3a00a", size = 78888, upload-time = "2025-06-12T22:23:04.272Z" }, +] + [[package]] name = "pathspec" version = "1.1.1" @@ -5013,6 +5224,23 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/bc/60/5382c03e1970de634027cee8e1b7d39776b778b81812aaf45b694dfe9e28/pillow-12.2.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:bfa9c230d2fe991bed5318a5f119bd6780cda2915cca595393649fc118ab895e", size = 7080946, upload-time = "2026-04-01T14:46:11.734Z" }, ] +[[package]] +name = "piper-plus-g2p" +version = "0.2.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/3f/93/60719a594ec396432df84de669fceac26bf7b8220aa5d9538ad5f620f914/piper_plus_g2p-0.2.0.tar.gz", hash = "sha256:e99c9c7c81660ea6146542b1f035c41aeb16e9eb958f3a9b01daf64b97e19165", size = 145250, upload-time = "2026-04-07T08:51:32.458Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e4/d4/c3fb6dd8e5481452d0943713460d9c6702198b3e0503cd6a04dd0bb0ac24/piper_plus_g2p-0.2.0-py3-none-any.whl", hash = "sha256:151662d920f27f6d674f073fabd386ec17f60aa5ab7f7a35193ba6c8a444599d", size = 68393, upload-time = "2026-04-07T08:51:31.031Z" }, +] + +[package.optional-dependencies] +all = [ + { name = "g2p-en", marker = "python_full_version >= '3.11'" }, + { name = "g2pk2", marker = "python_full_version >= '3.11'" }, + { name = "pyopenjtalk-plus", marker = "python_full_version >= '3.11'" }, + { name = "pypinyin", marker = "python_full_version >= '3.11'" }, +] + [[package]] name = "platformdirs" version = "4.10.0" @@ -5696,6 +5924,42 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/b2/e6/94145d714402fd5ade00b5661f2d0ab981219e07f7db9bfa16786cdb9c04/pynndescent-0.6.0-py3-none-any.whl", hash = "sha256:dc8c74844e4c7f5cbd1e0cd6909da86fdc789e6ff4997336e344779c3d5538ef", size = 73511, upload-time = "2026-01-08T21:29:57.306Z" }, ] +[[package]] +name = "pyopenjtalk-plus" +version = "0.4.1.post9" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "numpy", version = "1.26.4", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.11' and platform_machine == 'x86_64' and sys_platform == 'darwin'" }, + { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "(python_full_version >= '3.11' and platform_machine != 'x86_64') or (python_full_version >= '3.11' and sys_platform != 'darwin')" }, + { name = "pydantic", marker = "python_full_version >= '3.11'" }, + { name = "sudachidict-core", marker = "python_full_version >= '3.11'" }, + { name = "sudachipy", marker = "python_full_version >= '3.11'" }, + { name = "typing-extensions", marker = "python_full_version >= '3.11'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/1a/e7/03cc1d971260ae90cc35370965d6d8e612cc8b8588d2f81db7cd2accf9ad/pyopenjtalk_plus-0.4.1.post9.tar.gz", hash = "sha256:cdcb0746659857554c6dad23956cad77e21f76c9f3dfa000ea2f8d4f0ba11d99", size = 25023543, upload-time = "2026-08-11T21:13:31.021Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fc/b6/faca99e7015b81706f819fb7239e1e1d73d941ba6c9b420e63d6a245a911/pyopenjtalk_plus-0.4.1.post9-cp310-cp310-macosx_10_9_universal2.whl", hash = "sha256:1f0de887a624b86109464d9ec819e705d7412ebeda87a6abc0f828fc48091932", size = 25654409, upload-time = "2026-08-11T21:12:00.274Z" }, + { url = "https://files.pythonhosted.org/packages/e7/32/ebfe5a5153ec7fb468142ed674418ba1381afcb2f50d5fb3f6f2e64e569e/pyopenjtalk_plus-0.4.1.post9-cp310-cp310-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:5b0e6ddebf2cf60ef480e1a9861ff2f08e6b04e70fe3ca62f349f601af6adb5b", size = 31198254, upload-time = "2026-08-11T21:12:05.602Z" }, + { url = "https://files.pythonhosted.org/packages/0b/4d/a95e20214716b07ba89260c24bab157ff3f3b91a0a41c5e04ff103a2d844/pyopenjtalk_plus-0.4.1.post9-cp310-cp310-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5acf59e7a0750a5a35e1a76a724f257e995a71f8c06ff945f8436b8f9b62b5b9", size = 31281542, upload-time = "2026-08-11T21:12:10.513Z" }, + { url = "https://files.pythonhosted.org/packages/40/02/939abaa07066337ce1a9e405159f186b2b383fd4ab89754c68af6af29026/pyopenjtalk_plus-0.4.1.post9-cp310-cp310-win_amd64.whl", hash = "sha256:c2624815bef0528628bbdc52dec298b0eb577df261f19088b7bb5018c5ca0b51", size = 25064332, upload-time = "2026-08-11T21:12:14.817Z" }, + { url = "https://files.pythonhosted.org/packages/02/2c/8dacb54db5148d65802d2932fae702227e7ebbc4e8578ba417da872cc61d/pyopenjtalk_plus-0.4.1.post9-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:5e91f4d43740e550bc8a1fb8923fd5a490f06f5fbcac0fba9280e3b7d7b954e9", size = 25652290, upload-time = "2026-08-11T21:12:19.479Z" }, + { url = "https://files.pythonhosted.org/packages/a1/29/e55ccaeffb6746aa5c5d9f5365472c8899e21e788f33a7e4d8156045aa73/pyopenjtalk_plus-0.4.1.post9-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f35a6b1f977347c8b6b233728bc971fa6561cbcb1a23f6c94a79a8c763d64401", size = 31298276, upload-time = "2026-08-11T21:12:24.419Z" }, + { url = "https://files.pythonhosted.org/packages/fa/71/e6937ce16f4d1c35a161e463e0176220a3147ee214f9f56ebc0b1632cc60/pyopenjtalk_plus-0.4.1.post9-cp311-cp311-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b3a991396d641cf5983bf3c49e932d6f878c8a12215cad64871c82354d22cd78", size = 31388339, upload-time = "2026-08-11T21:12:29.587Z" }, + { url = "https://files.pythonhosted.org/packages/18/df/50e8185a5a5d441e404af7c0407491bc92ca76d44ceaf8cd086066365184/pyopenjtalk_plus-0.4.1.post9-cp311-cp311-win_amd64.whl", hash = "sha256:58cda3998e1c8f4d820e07757803e8cd9ad324c359af59f398c30ba38fb88865", size = 25065816, upload-time = "2026-08-11T21:12:34.144Z" }, + { url = "https://files.pythonhosted.org/packages/49/46/f509711876c6997842586bd7a55f8803dc0d37d8aa2ab65485d36b844507/pyopenjtalk_plus-0.4.1.post9-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:3a0c9e3d7a61e72be10c6e2dc37449037e72b2b6c7a97633eb03e393836936a2", size = 25656701, upload-time = "2026-08-11T21:12:38.91Z" }, + { url = "https://files.pythonhosted.org/packages/e1/ce/d1d01cbdc4b2ba9b01919f6f017856997d8fe4fac035e8e0fd08923384b5/pyopenjtalk_plus-0.4.1.post9-cp312-cp312-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f98993e1e943d4313e4113992b137714c0ac8f6a383d1fd89747a592222bdddf", size = 31274657, upload-time = "2026-08-11T21:12:43.76Z" }, + { url = "https://files.pythonhosted.org/packages/81/37/aa7b83620b5aadd7f47c42d0f04a2cad535e7fece52cc0c2387b97db3e85/pyopenjtalk_plus-0.4.1.post9-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:856b1ee7be326015490c23fc55046688492e2d99554852e476a24a10c8e77ba0", size = 31362445, upload-time = "2026-08-11T21:12:48.526Z" }, + { url = "https://files.pythonhosted.org/packages/9c/f3/bf7c62d0c0e8543a6b93d18df4527a0f060b5a21dd7e64408fa16e98ebad/pyopenjtalk_plus-0.4.1.post9-cp312-cp312-win_amd64.whl", hash = "sha256:7c13d8370d82e625c3ac0712d813da498c8f145f6ff246aec482753db952a5a9", size = 25059224, upload-time = "2026-08-11T21:12:52.391Z" }, + { url = "https://files.pythonhosted.org/packages/28/2b/3b77d3b18808f719106988df846956762026150447f6bc4e37372a195294/pyopenjtalk_plus-0.4.1.post9-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:7122013811e02ac7c2676c4bd7e5c2d37eb6f02b11944ff1fdf3fa4667d4cd5c", size = 25655431, upload-time = "2026-08-11T21:12:56.419Z" }, + { url = "https://files.pythonhosted.org/packages/13/27/ebf087640df43f6f319cde946a69dba082b1ecb82f56b8106e62e09c4c05/pyopenjtalk_plus-0.4.1.post9-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:10d3b80ac91f193861909f08b945b11e037965f6eadb767529b975465382282c", size = 31267213, upload-time = "2026-08-11T21:13:00.516Z" }, + { url = "https://files.pythonhosted.org/packages/22/d9/19994eae3451aafdff7d24ef73d06c259d310c8fe5b091725d42b291e234/pyopenjtalk_plus-0.4.1.post9-cp313-cp313-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:f7c1048013e8ce100f4f9d489d2acfc603756fb1a62ca0a2310475bc7e4fa365", size = 31357892, upload-time = "2026-08-11T21:13:04.907Z" }, + { url = "https://files.pythonhosted.org/packages/a5/ea/f5f4c9d1dd9f1788e9bf22a05ed28e6aa15698decda247bf1563610a5538/pyopenjtalk_plus-0.4.1.post9-cp313-cp313-win_amd64.whl", hash = "sha256:f3699de7491d958685ffd2656cb0f3ac1d5c1ca469f3da7caf2e5aa4617ef3b1", size = 25059130, upload-time = "2026-08-11T21:13:09.144Z" }, + { url = "https://files.pythonhosted.org/packages/de/d1/04c4a08db318048ca632994bcae2b46a5401dd25b1ea53effd19aacf642d/pyopenjtalk_plus-0.4.1.post9-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:75d1a93e97fe627ee88a5cc9c03af48cef440a957e66aab801e4be0bc1972384", size = 25656908, upload-time = "2026-08-11T21:13:13.148Z" }, + { url = "https://files.pythonhosted.org/packages/81/e5/8d555862f0f7f929aab041fb51d1afc6240552bbcbd2d6cf4af0a1e3ed9d/pyopenjtalk_plus-0.4.1.post9-cp314-cp314-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d669b461203e453698da8f3a1772d15802e64cc4f0d0278d65223d30339d0302", size = 31262061, upload-time = "2026-08-11T21:13:17.638Z" }, + { url = "https://files.pythonhosted.org/packages/ec/de/2cb80918d0a7b7f0260a97203f383ed79cc222b28e5b1cd95eed10b17f84/pyopenjtalk_plus-0.4.1.post9-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:71d06f9318d6fcdf63ba61549ea22765d674a8cfb8a29984b73915cd7cf40486", size = 31344566, upload-time = "2026-08-11T21:13:22.131Z" }, + { url = "https://files.pythonhosted.org/packages/0f/0d/a8542f269f7e912fb2920fa7d1eaf8e88c36bd1ea915c63c350fc5378f45/pyopenjtalk_plus-0.4.1.post9-cp314-cp314-win_amd64.whl", hash = "sha256:273a7829ebbc4ddfa94c11bdf47d0aac89b418a144afb6b9d80c745416e47578", size = 25745062, upload-time = "2026-08-11T21:13:26.909Z" }, +] + [[package]] name = "pyparsing" version = "3.3.2" @@ -5705,6 +5969,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/10/bd/c038d7cc38edc1aa5bf91ab8068b63d4308c66c4c8bb3cbba7dfbc049f9c/pyparsing-3.3.2-py3-none-any.whl", hash = "sha256:850ba148bd908d7e2411587e247a1e4f0327839c40e2e5e6d05a007ecc69911d", size = 122781, upload-time = "2026-01-21T03:57:55.912Z" }, ] +[[package]] +name = "pypinyin" +version = "0.55.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/b4/a4/784cf98c09e0dc22776b0d7d8a4a5b761218bcae4608c2416ce1e167c8af/pypinyin-0.55.0.tar.gz", hash = "sha256:b5711b3a0c6f76e67408ec6b2e3c4987a3a806b7c528076e7c7b86fcf0eaa66b", size = 839836, upload-time = "2025-07-20T12:01:50.657Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b9/7b/4cabc76fcc21c3c7d5c671d8783984d30ac9d3bb387c4ba784fca3cdfa3a/pypinyin-0.55.0-py2.py3-none-any.whl", hash = "sha256:d53b1e8ad2cdb815fb2cb604ed3123372f5a28c6f447571244aca36fc62a286f", size = 840203, upload-time = "2025-07-20T12:01:48.535Z" }, +] + [[package]] name = "pyproject-hooks" version = "1.2.0" @@ -5766,6 +6039,41 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/0b/d7/1959b9648791274998a9c3526f6d0ec8fd2233e4d4acce81bbae76b44b2a/python_dotenv-1.2.2-py3-none-any.whl", hash = "sha256:1d8214789a24de455a8b8bd8ae6fe3c6b69a5e3d64aa8a8e5d68e694bbcb285a", size = 22101, upload-time = "2026-03-01T16:00:25.09Z" }, ] +[[package]] +name = "python-mecab-ko" +version = "1.3.7" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "python-mecab-ko-dic", marker = "python_full_version >= '3.11' and sys_platform != 'win32'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/5b/22/81584e36d231b009075b6440da0eda4400f7293f853097b028263074f280/python_mecab_ko-1.3.7.tar.gz", hash = "sha256:69cbb2ac559a3169c22b1a3aa5d3c247d2f7902d9fe7dc9966189a9c7694af0b", size = 13830, upload-time = "2024-07-14T16:06:24.803Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/56/72/7ba507bc941f50bbbfafa96314fc145967b91387f34fd1c4e54af9de98fe/python_mecab_ko-1.3.7-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:4760efe6327b5707f55db2b4a6f8fb047fe8e068577a9a913304bb0d12e7de44", size = 413160, upload-time = "2024-07-14T16:05:13.231Z" }, + { url = "https://files.pythonhosted.org/packages/b9/4d/3cf853f6c8abdb901a45b2478f4915c0df387b06b66057196c792f287050/python_mecab_ko-1.3.7-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:27a03ae50aabc7f057c26ad5e4c6c4d431cf696778e45025e208d2f6b7bf115d", size = 349152, upload-time = "2024-07-14T16:05:15.38Z" }, + { url = "https://files.pythonhosted.org/packages/da/18/b685841eb245560903732b50f2a45c488452ad4162a01320acedfc622c4a/python_mecab_ko-1.3.7-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:d8d2539e7ea91eb0705381f75e64c626be4eba69824a8c82fbdf2c4e48a1d389", size = 558829, upload-time = "2024-07-14T16:05:17.263Z" }, + { url = "https://files.pythonhosted.org/packages/c3/a9/b121faaa74c41adc9f5f9c1862e5ba6873886610fcae8328ddd5386f2a39/python_mecab_ko-1.3.7-cp310-cp310-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:c2bad59670b280548b9060c1b511f6f088c09b977355de7192e9d0044b8f724b", size = 599923, upload-time = "2024-07-14T16:05:19.327Z" }, + { url = "https://files.pythonhosted.org/packages/b9/e0/b0be8e17b4ace180a323bb8ddf6213a6e984c18c403c65c6e71e88c4cb4f/python_mecab_ko-1.3.7-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:e8c4347f075b8748cbc5695f6b91120b0e388344eab5d9c26d50ad3c57c35754", size = 577095, upload-time = "2024-07-14T16:05:21.431Z" }, + { url = "https://files.pythonhosted.org/packages/c6/77/6d6bffa6217550d1dde3c421e40f422c924be1ed82effa1012cf7d24d215/python_mecab_ko-1.3.7-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:da1cc9de07e75beb2d4067c1c072ecabdb293440633fc0e32f2875a14e703829", size = 414576, upload-time = "2024-07-14T16:05:28.408Z" }, + { url = "https://files.pythonhosted.org/packages/06/a6/ab4405b9228e62ced5d7134e202ec8970296d58ceca5e991c76abbe04701/python_mecab_ko-1.3.7-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:523153de14262c413838852742541d48ad99d41ab8f6c5413a226319ee4c25ef", size = 350486, upload-time = "2024-07-14T16:05:30.115Z" }, + { url = "https://files.pythonhosted.org/packages/e5/73/6768de76a1b0b1120b07a58334b89db214d740d62504a3fe691fd8738df4/python_mecab_ko-1.3.7-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:010ca2297e63d08a772466dd401d36ed9914502b8794c08948427a4083b3202c", size = 560126, upload-time = "2024-07-14T16:05:31.331Z" }, + { url = "https://files.pythonhosted.org/packages/5c/7d/28552e79b7bd4bc9d39b0bf1a037f541c5b2cf07467c1a274715b1afecde/python_mecab_ko-1.3.7-cp311-cp311-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:d7b35116e98fb736f7c9550eb1a74cfb6aa35c39b0b43cbe7a8837bfa3cd39d4", size = 602904, upload-time = "2024-07-14T16:05:32.844Z" }, + { url = "https://files.pythonhosted.org/packages/c4/5e/2986910cb757eb49f592b33977be05396ebeeb6e18ec6a4643f6c9fdce2e/python_mecab_ko-1.3.7-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0933d3fcb84f6ed36cce49f1939604ac0fcaf4460441e832cb98ca1bdce74a37", size = 580930, upload-time = "2024-07-14T16:05:34.825Z" }, + { url = "https://files.pythonhosted.org/packages/c2/7a/980c48f19367353da0d0ea4ffc5c73be0893987bc4c8bf4522391b4e0b14/python_mecab_ko-1.3.7-cp312-cp312-macosx_10_9_x86_64.whl", hash = "sha256:7721f69381dac572a1598e5906cc5faba233ed48bc6ff8672082a519d7db0ba1", size = 414066, upload-time = "2024-07-14T16:05:41.139Z" }, + { url = "https://files.pythonhosted.org/packages/ee/3d/8c13f7d6cbcc9b84b1797552de60fc2ff10012d801598b75aa41ff17db41/python_mecab_ko-1.3.7-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:eae5eb6178b06019e3773e9dde126dd29df5ed417406be5611ebdd0f8839c1e1", size = 349619, upload-time = "2024-07-14T16:05:42.786Z" }, + { url = "https://files.pythonhosted.org/packages/e7/01/3a4032f78370375d79b7f6b7eaa1602a4070d28dee8f7f3dfa8c87e73940/python_mecab_ko-1.3.7-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:8e90e8c1009f8f6aa0dfc43c916ff481dc79aa5a7e528a41a193add9c61ac6d1", size = 560604, upload-time = "2024-07-14T16:05:44.062Z" }, + { url = "https://files.pythonhosted.org/packages/f4/cb/169c68047a569e2e4f2a2243c4576df72255d1a4ab0b5a286ff75bc41a76/python_mecab_ko-1.3.7-cp312-cp312-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:a205ca4da908df39d6d70f968426d0e9dc79274a6d34b13a5588ab52f0e12be8", size = 602363, upload-time = "2024-07-14T16:05:45.667Z" }, + { url = "https://files.pythonhosted.org/packages/aa/88/dd2e5e7d44351bba0c9fb462273df9dcd43c3e000bfd0a4279e52584e2f7/python_mecab_ko-1.3.7-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3c0e98a7d94278f4f5d93f03e35cc8044460c0076ab4698b764d5c44bd897dbe", size = 579619, upload-time = "2024-07-14T16:05:47.166Z" }, +] + +[[package]] +name = "python-mecab-ko-dic" +version = "2.1.1.post2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/b5/cc/0188ca47e3d508c0961fb915ab9c71fd6facc821afb906e7b9080009d4ec/python-mecab-ko-dic-2.1.1.post2.tar.gz", hash = "sha256:2c423713bdc475345ec98cd084b30759458f8f06c38a9ef94ab8687942c2cd34", size = 34179115, upload-time = "2022-12-14T08:41:53.468Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/4e/63/23fb02dd36d96527eb0c93300e8db22ab42805c1a232af621c5a8f175e37/python_mecab_ko_dic-2.1.1.post2-py3-none-any.whl", hash = "sha256:ef8f4e80c8976f1340a7264abb0c96f384fe059fd897584aeba0151753c6ae9b", size = 34457665, upload-time = "2022-12-14T08:41:39.777Z" }, +] + [[package]] name = "python-multipart" version = "0.0.29" @@ -7291,6 +7599,37 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/9f/85/492183764d5d01d4514be3730fdb8e228a80605783099551c51627578b5d/starlette-1.2.0-py3-none-any.whl", hash = "sha256:36e0c76ac59157e75dc4b3bdeafba97fb04eaf1878045f15dbef666a6f092ed7", size = 73213, upload-time = "2026-05-28T11:42:48.801Z" }, ] +[[package]] +name = "sudachidict-core" +version = "20260723.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "sudachipy", marker = "python_full_version >= '3.11'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/9c/b5/ddf1c54817572a24ee4172a6290dc574146f9049b2330fac112c25e00f0d/sudachidict_core-20260723.1.tar.gz", hash = "sha256:06605ab827f0ea3f37aaa5dee7fa2e5324f0a2d5bc7662e78a0a597b7eac7955", size = 9144, upload-time = "2026-09-18T08:52:32.613Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/85/af/ba8419f684865b8cca587e01cedb41ba83fbdc985d75ab9e6ff38fdedf1a/sudachidict_core-20260723.1-py3-none-any.whl", hash = "sha256:2b711055dca03423869e491eca0ddbe3e17c4d7418ed738fd7c75d4e0eb9e4b1", size = 76938227, upload-time = "2026-09-24T00:25:03.364Z" }, +] + +[[package]] +name = "sudachipy" +version = "0.7.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/fc/1d/b34702f38a8bd34972836538d222209095d5aba7faff79d56bce3fb2640c/sudachipy-0.7.0.tar.gz", hash = "sha256:14a2505e7d086fe9da64c0048d8e2de13cdded3aac4f1e047929cef51dba338c", size = 304664, upload-time = "2026-09-24T01:34:56.238Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/44/aa/47edbec02618803ac8ef8f1561b233760bbd266fb040ba9d4d577a5f1773/sudachipy-0.7.0-cp310-abi3-macosx_10_12_universal2.whl", hash = "sha256:6be34e856c99498e990904f7155e1a102f675cf5edfdaf7a396f26d71ce4b333", size = 3147103, upload-time = "2026-09-24T01:34:43.053Z" }, + { url = "https://files.pythonhosted.org/packages/d7/b3/2117420a05b2785ce561912baff01dd7872f5bca6b60dce49dc592a9f600/sudachipy-0.7.0-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:907eadc1db305f9bc6e502a9acf9ff1ac871cd8d4ac3a2e0aaba3ddc60e93b2a", size = 1622350, upload-time = "2026-09-24T01:34:44.636Z" }, + { url = "https://files.pythonhosted.org/packages/88/11/3b740eea1796d1ce6f7bdaa1f5470685dbcbfcacf751f3aa06d9da72856f/sudachipy-0.7.0-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:0e60cefd3a7a9680206ae160bf86d266d0d60e434adac2384f8c7a6dfb0f3ba7", size = 1561984, upload-time = "2026-09-24T01:34:45.761Z" }, + { url = "https://files.pythonhosted.org/packages/9f/12/5455c3e4afea272c1ef0995c489dc9c2a6ead2a6d980822a53ae879a2042/sudachipy-0.7.0-cp310-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:68be139f5d7f053eba16fd3f792e4ef0c6bfed81a1b63e68b4544c7159296034", size = 1720464, upload-time = "2026-09-24T01:34:46.92Z" }, + { url = "https://files.pythonhosted.org/packages/1d/2d/b2c2511b3627032e81ef1626d6bce4506efacc7b698ca0e70cf6b4345e86/sudachipy-0.7.0-cp310-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:dfc90631aa2276d2165e1995bc94b56c9f5bc27b875f43fb7d67e4a3b7346d4f", size = 1758941, upload-time = "2026-09-24T01:34:47.986Z" }, + { url = "https://files.pythonhosted.org/packages/e9/a1/03e40b0eadbdd48ed47aef7aaf27d4703f5cc225750872af1a26de08ec2b/sudachipy-0.7.0-cp310-abi3-win_amd64.whl", hash = "sha256:8d0429db2d02d408daa7dccd062a6d8c2984a07f8e448b49a3ea694377617658", size = 1515929, upload-time = "2026-09-24T01:34:49.307Z" }, + { url = "https://files.pythonhosted.org/packages/f2/f7/580ece6a559e0cad8a9a3fff36bbd478e335e125c96c1910abe20f3b6bc2/sudachipy-0.7.0-cp314-cp314t-macosx_10_15_universal2.whl", hash = "sha256:1bf2254ee6357e0defa498630f188fefa8a543b82887077cfb1596bc3e8d4840", size = 3139240, upload-time = "2026-09-24T01:34:50.418Z" }, + { url = "https://files.pythonhosted.org/packages/62/20/4475da1e0393bef7d136f91214d560730006d2a4792e585003b17f2779e1/sudachipy-0.7.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:efd56375584dec9523fe9a26daa69da02a70ae2f2c22d25248e65a4998cae8e1", size = 1626313, upload-time = "2026-09-24T01:34:51.665Z" }, + { url = "https://files.pythonhosted.org/packages/8b/75/952595fe7fa84a8f90cbe8f94a7cb5210642f166ec0ec1d7d808abef978e/sudachipy-0.7.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:9446eba17985b1d14926afc6312fb0367687ab1b5397fa65d7e2cdd1745451c9", size = 1549814, upload-time = "2026-09-24T01:34:52.708Z" }, + { url = "https://files.pythonhosted.org/packages/cf/39/5e138a2f949bf3af45951980382adc69d0d975448e604dc1801325f0008a/sudachipy-0.7.0-cp314-cp314t-manylinux_2_28_aarch64.whl", hash = "sha256:f1e24f71a71816dcbc6fb759a5819cca5d603a4d49b7ade63dd0fe49468ed22c", size = 1712756, upload-time = "2026-09-24T01:34:53.813Z" }, + { url = "https://files.pythonhosted.org/packages/2d/f4/2df0f7a5dce327d82c451be96151afa8f7e3324a3a8d71845eec5be26f3e/sudachipy-0.7.0-cp314-cp314t-manylinux_2_28_x86_64.whl", hash = "sha256:c8306eab503d891e394fb21156e417c2f9f54cc39f56dddb8cf2b40620d470fa", size = 1750689, upload-time = "2026-09-24T01:34:55.08Z" }, +] + [[package]] name = "sympy" version = "1.14.0" @@ -7911,6 +8250,12 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/1b/98/f63318ccbe75c810011fe9233884c5d348d94d90005de1b79e5f93bef9c0/umap_learn-0.5.12-py3-none-any.whl", hash = "sha256:f2a85d2a2adcb52b541bed9b27a23ca169b56bb1b23283abeebfb8dfb8a42fe5", size = 91849, upload-time = "2026-04-08T20:03:52.561Z" }, ] +[[package]] +name = "unicodecsv" +version = "0.14.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/6f/a4/691ab63b17505a26096608cc309960b5a6bdf39e4ba1a793d5f9b1a53270/unicodecsv-0.14.1.tar.gz", hash = "sha256:018c08037d48649a0412063ff4eda26eaa81eff1546dbffa51fa5293276ff7fc", size = 10267, upload-time = "2015-09-22T22:00:19.516Z" } + [[package]] name = "urllib3" version = "2.7.0"