mirror of
https://github.com/volcengine/OpenViking.git
synced 2026-10-02 02:07:33 +08:00
Revert audio parser changes from this branch
The audio parser feature is unrelated to memory health stats and belongs in its own PR (#707). Reverts audio.py to pre-rewrite state, removes the unused audio_summary.yaml template and audio parser tests. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.6
parent
9d63ee5547
commit
898cc8e426
@@ -1,117 +1,41 @@
|
||||
# Copyright (c) 2026 Beijing Volcano Engine Technology Co., Ltd.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""
|
||||
Audio parser with metadata extraction and Whisper transcription.
|
||||
Audio parser - Future implementation.
|
||||
|
||||
Features:
|
||||
1. Speech-to-text transcription using Whisper API
|
||||
2. Audio metadata extraction (duration, sample rate, channels) via mutagen
|
||||
3. Timestamp alignment for transcribed text
|
||||
4. Generate structured ResourceNode with transcript segments
|
||||
Planned Features:
|
||||
1. Speech-to-text transcription using ASR models
|
||||
2. Audio metadata extraction (duration, sample rate, channels)
|
||||
3. Speaker diarization (identify different speakers)
|
||||
4. Timestamp alignment for transcribed text
|
||||
5. Generate structured ResourceNode with transcript
|
||||
|
||||
Supported formats: MP3, WAV, OGG, FLAC, AAC, M4A, OPUS
|
||||
Example workflow:
|
||||
1. Load audio file
|
||||
2. Extract metadata (duration, format, sample rate)
|
||||
3. Transcribe speech to text using Whisper or similar
|
||||
4. (Optional) Perform speaker diarization
|
||||
5. Create ResourceNode with:
|
||||
- type: NodeType.ROOT
|
||||
- children: sections for each speaker/timestamp
|
||||
- meta: audio metadata and timestamps
|
||||
6. Return ParseResult
|
||||
|
||||
Supported formats: MP3, WAV, OGG, FLAC, AAC, M4A
|
||||
"""
|
||||
|
||||
import io
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Union
|
||||
from typing import List, Optional, Union
|
||||
|
||||
from openviking.parse.base import NodeType, ParseResult, ResourceNode
|
||||
from openviking.parse.parsers.base_parser import BaseParser
|
||||
from openviking.parse.parsers.media.constants import AUDIO_EXTENSIONS
|
||||
from openviking_cli.utils.config.parser_config import AudioConfig
|
||||
from openviking_cli.utils.logger import get_logger
|
||||
|
||||
logger = get_logger(__name__)
|
||||
|
||||
# Magic bytes for audio format validation
|
||||
AUDIO_MAGIC_BYTES: Dict[str, List[bytes]] = {
|
||||
".mp3": [b"ID3", b"\xff\xfb", b"\xff\xf3", b"\xff\xf2"],
|
||||
".wav": [b"RIFF"],
|
||||
".ogg": [b"OggS"],
|
||||
".flac": [b"fLaC"],
|
||||
".aac": [b"\xff\xf1", b"\xff\xf9"],
|
||||
".m4a": [b"\x00\x00\x00", b"ftypM4A", b"ftypisom"],
|
||||
".opus": [b"OggS"],
|
||||
}
|
||||
|
||||
|
||||
def _try_import_mutagen():
|
||||
"""Lazily import mutagen, returning None if not installed."""
|
||||
try:
|
||||
import mutagen
|
||||
|
||||
return mutagen
|
||||
except ImportError:
|
||||
return None
|
||||
|
||||
|
||||
def _format_timestamp(seconds: float) -> str:
|
||||
"""Format seconds as MM:SS or H:MM:SS."""
|
||||
hours = int(seconds // 3600)
|
||||
minutes = int((seconds % 3600) // 60)
|
||||
secs = int(seconds % 60)
|
||||
if hours > 0:
|
||||
return f"{hours}:{minutes:02d}:{secs:02d}"
|
||||
return f"{minutes}:{secs:02d}"
|
||||
|
||||
|
||||
def _extract_metadata_mutagen(file_path: Path) -> Dict[str, Any]:
|
||||
"""
|
||||
Extract audio metadata using mutagen.
|
||||
|
||||
Args:
|
||||
file_path: Path to audio file
|
||||
|
||||
Returns:
|
||||
Dictionary with duration, sample_rate, channels, bitrate, format
|
||||
"""
|
||||
mutagen = _try_import_mutagen()
|
||||
if mutagen is None:
|
||||
logger.warning(
|
||||
"[AudioParser] mutagen not installed, skipping metadata extraction. "
|
||||
"Install with: pip install mutagen"
|
||||
)
|
||||
return {}
|
||||
|
||||
try:
|
||||
audio = mutagen.File(str(file_path))
|
||||
if audio is None:
|
||||
logger.warning(f"[AudioParser] mutagen could not identify file: {file_path}")
|
||||
return {}
|
||||
|
||||
meta: Dict[str, Any] = {}
|
||||
|
||||
# Duration
|
||||
if hasattr(audio.info, "length"):
|
||||
meta["duration"] = round(audio.info.length, 2)
|
||||
|
||||
# Sample rate
|
||||
if hasattr(audio.info, "sample_rate"):
|
||||
meta["sample_rate"] = audio.info.sample_rate
|
||||
|
||||
# Channels
|
||||
if hasattr(audio.info, "channels"):
|
||||
meta["channels"] = audio.info.channels
|
||||
|
||||
# Bitrate (bits per second)
|
||||
if hasattr(audio.info, "bitrate"):
|
||||
meta["bitrate"] = audio.info.bitrate
|
||||
|
||||
return meta
|
||||
|
||||
except Exception as e:
|
||||
logger.warning(f"[AudioParser] mutagen metadata extraction failed: {e}")
|
||||
return {}
|
||||
|
||||
|
||||
class AudioParser(BaseParser):
|
||||
"""
|
||||
Audio parser for audio files.
|
||||
|
||||
Extracts metadata via mutagen and transcribes speech via Whisper API.
|
||||
Falls back to metadata-only output when transcription is unavailable.
|
||||
"""
|
||||
|
||||
def __init__(self, config: Optional[AudioConfig] = None, **kwargs):
|
||||
@@ -129,28 +53,23 @@ class AudioParser(BaseParser):
|
||||
"""Return supported audio file extensions."""
|
||||
return AUDIO_EXTENSIONS
|
||||
|
||||
async def parse(
|
||||
self, source: Union[str, Path], instruction: str = "", **kwargs
|
||||
) -> ParseResult:
|
||||
async def parse(self, source: Union[str, Path], instruction: str = "", **kwargs) -> ParseResult:
|
||||
"""
|
||||
Parse audio file - extract metadata, transcribe via Whisper, build ResourceNode tree.
|
||||
Parse audio file - only copy original file and extract basic metadata, no content understanding.
|
||||
|
||||
Args:
|
||||
source: Audio file path
|
||||
instruction: Processing instruction
|
||||
**kwargs: Additional parsing parameters
|
||||
|
||||
Returns:
|
||||
ParseResult with audio content tree
|
||||
ParseResult with audio content
|
||||
|
||||
Raises:
|
||||
FileNotFoundError: If source file does not exist
|
||||
ValueError: If file signature does not match expected format
|
||||
IOError: If audio processing fails
|
||||
"""
|
||||
from openviking.storage.viking_fs import get_viking_fs
|
||||
|
||||
start_time = time.monotonic()
|
||||
|
||||
# Convert to Path object
|
||||
file_path = Path(source) if isinstance(source, str) else source
|
||||
if not file_path.exists():
|
||||
@@ -159,339 +78,160 @@ class AudioParser(BaseParser):
|
||||
viking_fs = get_viking_fs()
|
||||
temp_uri = viking_fs.create_temp_uri()
|
||||
|
||||
# Read audio bytes
|
||||
# Phase 1: Generate temporary files
|
||||
audio_bytes = file_path.read_bytes()
|
||||
ext = file_path.suffix
|
||||
|
||||
# Validate magic bytes
|
||||
self._validate_audio_bytes(audio_bytes, ext, file_path)
|
||||
|
||||
from openviking_cli.utils.uri import VikingURI
|
||||
|
||||
# Sanitize original filename (replace spaces with underscores)
|
||||
original_filename = file_path.name.replace(" ", "_")
|
||||
# Root directory name: filename stem + _ + extension (without dot)
|
||||
stem = file_path.stem.replace(" ", "_")
|
||||
ext_no_dot = ext[1:] if ext else ""
|
||||
root_dir_name = VikingURI.sanitize_segment(f"{stem}_{ext_no_dot}")
|
||||
root_dir_uri = f"{temp_uri}/{root_dir_name}"
|
||||
await viking_fs.mkdir(root_dir_uri, exist_ok=True)
|
||||
|
||||
# Save original audio
|
||||
# 1.1 Save original audio with original filename (sanitized)
|
||||
await viking_fs.write_file_bytes(f"{root_dir_uri}/{original_filename}", audio_bytes)
|
||||
|
||||
# Extract metadata via mutagen
|
||||
mutagen_meta = _extract_metadata_mutagen(file_path)
|
||||
duration = mutagen_meta.get("duration", 0)
|
||||
sample_rate = mutagen_meta.get("sample_rate", 0)
|
||||
channels = mutagen_meta.get("channels", 0)
|
||||
bitrate = mutagen_meta.get("bitrate", 0)
|
||||
format_str = ext_no_dot.lower()
|
||||
|
||||
# Attempt transcription
|
||||
transcript_segments: List[Dict[str, Any]] = []
|
||||
full_transcript = ""
|
||||
warnings: List[str] = []
|
||||
|
||||
if self.config.enable_transcription:
|
||||
try:
|
||||
transcript_segments = await self._asr_transcribe_with_timestamps(
|
||||
audio_bytes, self.config.transcription_model, ext
|
||||
)
|
||||
if transcript_segments:
|
||||
full_transcript = "\n".join(
|
||||
seg["text"] for seg in transcript_segments
|
||||
)
|
||||
else:
|
||||
# Try plain transcription
|
||||
full_transcript = await self._asr_transcribe(
|
||||
audio_bytes, self.config.transcription_model, ext
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning(f"[AudioParser] Transcription failed: {e}")
|
||||
warnings.append(f"Transcription unavailable: {e}")
|
||||
|
||||
has_transcript = bool(full_transcript.strip())
|
||||
|
||||
# Save transcript file if available
|
||||
if has_transcript:
|
||||
transcript_md = self._build_transcript_markdown(
|
||||
transcript_segments, full_transcript, file_path.stem
|
||||
)
|
||||
await viking_fs.write_file(f"{root_dir_uri}/transcript.md", transcript_md)
|
||||
|
||||
# Build segment child nodes
|
||||
children = []
|
||||
if transcript_segments:
|
||||
for i, seg in enumerate(transcript_segments):
|
||||
seg_start = seg.get("start", 0)
|
||||
seg_end = seg.get("end", 0)
|
||||
seg_text = seg.get("text", "").strip()
|
||||
if not seg_text:
|
||||
continue
|
||||
|
||||
child = ResourceNode(
|
||||
type=NodeType.SECTION,
|
||||
title=f"segment_{i + 1:03d} ({_format_timestamp(seg_start)}-{_format_timestamp(seg_end)})",
|
||||
level=1,
|
||||
detail_file=None,
|
||||
content_path=None,
|
||||
children=[],
|
||||
content_type="text",
|
||||
meta={
|
||||
"start": seg_start,
|
||||
"end": seg_end,
|
||||
"text": seg_text,
|
||||
},
|
||||
)
|
||||
children.append(child)
|
||||
|
||||
# Build root node meta
|
||||
root_meta: Dict[str, Any] = {
|
||||
"duration": duration,
|
||||
"sample_rate": sample_rate,
|
||||
"channels": channels,
|
||||
"bitrate": bitrate,
|
||||
"format": format_str,
|
||||
"content_type": "audio",
|
||||
"source_title": file_path.stem,
|
||||
"semantic_name": file_path.stem,
|
||||
"original_filename": original_filename,
|
||||
"has_transcript": has_transcript,
|
||||
"segment_count": len(children),
|
||||
# 1.2 Validate audio file using magic bytes
|
||||
# Define magic bytes for supported audio formats
|
||||
audio_magic_bytes = {
|
||||
".mp3": [b"ID3", b"\xff\xfb", b"\xff\xf3", b"\xff\xf2"],
|
||||
".wav": [b"RIFF"],
|
||||
".ogg": [b"OggS"],
|
||||
".flac": [b"fLaC"],
|
||||
".aac": [b"\xff\xf1", b"\xff\xf9"],
|
||||
".m4a": [b"\x00\x00\x00", b"ftypM4A", b"ftypisom"],
|
||||
".opus": [b"OggS"],
|
||||
}
|
||||
|
||||
# Create root ResourceNode
|
||||
# Check magic bytes
|
||||
valid = False
|
||||
ext_lower = ext.lower()
|
||||
magic_list = audio_magic_bytes.get(ext_lower, [])
|
||||
for magic in magic_list:
|
||||
if len(audio_bytes) >= len(magic) and audio_bytes.startswith(magic):
|
||||
valid = True
|
||||
break
|
||||
|
||||
if not valid:
|
||||
raise ValueError(
|
||||
f"Invalid audio file: {file_path}. File signature does not match expected format {ext_lower}"
|
||||
)
|
||||
|
||||
# Extract audio metadata (placeholder)
|
||||
duration = 0
|
||||
sample_rate = 0
|
||||
channels = 0
|
||||
format_str = ext[1:].upper()
|
||||
|
||||
# Create ResourceNode - metadata only, no content understanding yet
|
||||
root_node = ResourceNode(
|
||||
type=NodeType.ROOT,
|
||||
title=file_path.stem,
|
||||
level=0,
|
||||
detail_file=None,
|
||||
content_path=None,
|
||||
children=children,
|
||||
content_type="audio",
|
||||
meta=root_meta,
|
||||
children=[],
|
||||
meta={
|
||||
"duration": duration,
|
||||
"sample_rate": sample_rate,
|
||||
"channels": channels,
|
||||
"format": format_str.lower(),
|
||||
"content_type": "audio",
|
||||
"source_title": file_path.stem,
|
||||
"semantic_name": file_path.stem,
|
||||
"original_filename": original_filename,
|
||||
},
|
||||
)
|
||||
|
||||
# Generate semantic info (L0 abstract, L1 overview)
|
||||
description = full_transcript if has_transcript else f"Audio file: {file_path.name}"
|
||||
await self._generate_semantic_info(
|
||||
root_node, description, viking_fs, has_transcript
|
||||
)
|
||||
|
||||
if not has_transcript:
|
||||
warnings.append(
|
||||
"No transcript available. Metadata-only output. "
|
||||
"Configure Whisper API or install openai-whisper for transcription."
|
||||
)
|
||||
|
||||
parse_time = time.monotonic() - start_time
|
||||
|
||||
# Phase 3: Build directory structure (handled by TreeBuilder)
|
||||
return ParseResult(
|
||||
root=root_node,
|
||||
source_path=str(file_path),
|
||||
temp_dir_path=temp_uri,
|
||||
source_format="audio",
|
||||
parser_name="AudioParser",
|
||||
parse_time=parse_time,
|
||||
meta={"content_type": "audio", "format": format_str},
|
||||
warnings=warnings,
|
||||
meta={"content_type": "audio", "format": format_str.lower()},
|
||||
)
|
||||
|
||||
def _validate_audio_bytes(
|
||||
self, audio_bytes: bytes, ext: str, file_path: Path
|
||||
) -> None:
|
||||
"""Validate audio file using magic bytes."""
|
||||
ext_lower = ext.lower()
|
||||
magic_list = AUDIO_MAGIC_BYTES.get(ext_lower, [])
|
||||
for magic in magic_list:
|
||||
if len(audio_bytes) >= len(magic) and audio_bytes.startswith(magic):
|
||||
return
|
||||
# If no magic bytes defined for this extension, skip validation
|
||||
if not magic_list:
|
||||
return
|
||||
raise ValueError(
|
||||
f"Invalid audio file: {file_path}. "
|
||||
f"File signature does not match expected format {ext_lower}"
|
||||
)
|
||||
|
||||
async def _asr_transcribe(
|
||||
self, audio_bytes: bytes, model: Optional[str], ext: str = ".mp3"
|
||||
) -> str:
|
||||
async def _asr_transcribe(self, audio_bytes: bytes, model: Optional[str]) -> str:
|
||||
"""
|
||||
Transcribe audio using Whisper API via OpenAI client.
|
||||
Generate audio transcription using ASR.
|
||||
|
||||
Args:
|
||||
audio_bytes: Audio binary data
|
||||
model: Whisper model name
|
||||
ext: File extension for mime type hint
|
||||
model: ASR model name
|
||||
|
||||
Returns:
|
||||
Transcription text
|
||||
Audio transcription in markdown format
|
||||
|
||||
TODO: Integrate with actual ASR API (Whisper, etc.)
|
||||
"""
|
||||
try:
|
||||
from openviking_cli.utils.config import get_openviking_config
|
||||
|
||||
config = get_openviking_config()
|
||||
import openai
|
||||
|
||||
client = openai.AsyncOpenAI(
|
||||
api_key=config.llm.api_key if hasattr(config, "llm") else None,
|
||||
)
|
||||
|
||||
audio_file = io.BytesIO(audio_bytes)
|
||||
audio_file.name = f"audio{ext}"
|
||||
|
||||
response = await client.audio.transcriptions.create(
|
||||
model=model or "whisper-1",
|
||||
file=audio_file,
|
||||
language=self.config.language,
|
||||
)
|
||||
|
||||
return response.text
|
||||
|
||||
except Exception as e:
|
||||
logger.warning(f"[AudioParser._asr_transcribe] Whisper API call failed: {e}")
|
||||
return ""
|
||||
# Fallback implementation - returns basic placeholder
|
||||
return "Audio transcription (ASR integration pending)\n\nThis is an audio. ASR transcription feature has not yet integrated external API."
|
||||
|
||||
async def _asr_transcribe_with_timestamps(
|
||||
self, audio_bytes: bytes, model: Optional[str], ext: str = ".mp3"
|
||||
) -> List[Dict[str, Any]]:
|
||||
self, audio_bytes: bytes, model: Optional[str]
|
||||
) -> Optional[str]:
|
||||
"""
|
||||
Transcribe audio with timestamps using Whisper API verbose_json format.
|
||||
Extract transcription with timestamps from audio using ASR.
|
||||
|
||||
Args:
|
||||
audio_bytes: Audio binary data
|
||||
model: Whisper model name
|
||||
ext: File extension
|
||||
model: ASR model name
|
||||
|
||||
Returns:
|
||||
List of segment dicts with keys: start, end, text
|
||||
Transcript with timestamps in markdown format, or None if not available
|
||||
|
||||
TODO: Integrate with ASR API
|
||||
"""
|
||||
try:
|
||||
from openviking_cli.utils.config import get_openviking_config
|
||||
|
||||
config = get_openviking_config()
|
||||
import openai
|
||||
|
||||
client = openai.AsyncOpenAI(
|
||||
api_key=config.llm.api_key if hasattr(config, "llm") else None,
|
||||
)
|
||||
|
||||
audio_file = io.BytesIO(audio_bytes)
|
||||
audio_file.name = f"audio{ext}"
|
||||
|
||||
response = await client.audio.transcriptions.create(
|
||||
model=model or "whisper-1",
|
||||
file=audio_file,
|
||||
response_format="verbose_json",
|
||||
timestamp_granularities=["segment"],
|
||||
language=self.config.language,
|
||||
)
|
||||
|
||||
segments = []
|
||||
if hasattr(response, "segments") and response.segments:
|
||||
for seg in response.segments:
|
||||
segments.append({
|
||||
"start": seg.get("start", 0) if isinstance(seg, dict) else getattr(seg, "start", 0),
|
||||
"end": seg.get("end", 0) if isinstance(seg, dict) else getattr(seg, "end", 0),
|
||||
"text": seg.get("text", "") if isinstance(seg, dict) else getattr(seg, "text", ""),
|
||||
})
|
||||
|
||||
return segments
|
||||
|
||||
except Exception as e:
|
||||
logger.warning(
|
||||
f"[AudioParser._asr_transcribe_with_timestamps] Whisper API call failed: {e}"
|
||||
)
|
||||
return []
|
||||
|
||||
def _build_transcript_markdown(
|
||||
self,
|
||||
segments: List[Dict[str, Any]],
|
||||
full_transcript: str,
|
||||
title: str,
|
||||
) -> str:
|
||||
"""
|
||||
Build a markdown transcript file from segments or plain text.
|
||||
|
||||
Args:
|
||||
segments: Timestamped transcript segments
|
||||
full_transcript: Full transcript text (used if no segments)
|
||||
title: Audio file title
|
||||
|
||||
Returns:
|
||||
Markdown-formatted transcript
|
||||
"""
|
||||
parts = [f"# Transcript: {title}\n"]
|
||||
|
||||
if segments:
|
||||
for seg in segments:
|
||||
start = _format_timestamp(seg.get("start", 0))
|
||||
end = _format_timestamp(seg.get("end", 0))
|
||||
text = seg.get("text", "").strip()
|
||||
if text:
|
||||
parts.append(f"**[{start} - {end}]** {text}\n")
|
||||
elif full_transcript.strip():
|
||||
parts.append(full_transcript.strip())
|
||||
parts.append("")
|
||||
|
||||
return "\n".join(parts)
|
||||
# Not implemented - return None
|
||||
return None
|
||||
|
||||
async def _generate_semantic_info(
|
||||
self,
|
||||
node: ResourceNode,
|
||||
description: str,
|
||||
viking_fs: Any,
|
||||
has_transcript: bool,
|
||||
) -> None:
|
||||
self, node: ResourceNode, description: str, viking_fs, has_transcript: bool
|
||||
):
|
||||
"""
|
||||
Generate L0 abstract and L1 overview for the audio resource.
|
||||
Phase 2: Generate abstract and overview.
|
||||
|
||||
Args:
|
||||
node: ResourceNode to update
|
||||
description: Audio transcript or description text
|
||||
description: Audio description
|
||||
viking_fs: VikingFS instance
|
||||
has_transcript: Whether transcript is available
|
||||
has_transcript: Whether transcript file exists
|
||||
"""
|
||||
# L0 abstract: short summary (< 256 chars)
|
||||
if has_transcript and len(description) > 50:
|
||||
first_sentence_end = description.find(".", 20)
|
||||
if 20 < first_sentence_end < 256:
|
||||
abstract = description[: first_sentence_end + 1]
|
||||
else:
|
||||
abstract = description[:253] + "..." if len(description) > 256 else description
|
||||
else:
|
||||
abstract = description[:253] + "..." if len(description) > 256 else description
|
||||
# Generate abstract (short summary, < 100 tokens)
|
||||
abstract = description[:200] if len(description) > 200 else description
|
||||
|
||||
# L1 overview
|
||||
# Generate overview (content summary + file list + usage instructions)
|
||||
overview_parts = [
|
||||
"## Content Summary\n",
|
||||
abstract,
|
||||
description,
|
||||
"\n\n## Available Files\n",
|
||||
(
|
||||
f"- {node.meta['original_filename']}: Original audio file "
|
||||
f"({node.meta['duration']}s, {node.meta['sample_rate']}Hz, "
|
||||
f"{node.meta['channels']}ch, {node.meta['format'].upper()} format)\n"
|
||||
),
|
||||
f"- {node.meta['original_filename']}: Original audio file ({node.meta['duration']}s, {node.meta['sample_rate']}Hz, {node.meta['channels']}ch, {node.meta['format'].upper()} format)\n",
|
||||
]
|
||||
|
||||
if has_transcript:
|
||||
overview_parts.append(
|
||||
"- transcript.md: Timestamped transcript from the audio\n"
|
||||
)
|
||||
overview_parts.append("- transcript.md: Transcript with timestamps from the audio\n")
|
||||
|
||||
overview_parts.append("\n## Usage\n")
|
||||
overview_parts.append("### Play Audio\n")
|
||||
overview_parts.append("```python\n")
|
||||
overview_parts.append("audio_bytes = await audio_resource.play()\n")
|
||||
overview_parts.append("# Returns: Audio file binary data\n")
|
||||
overview_parts.append("# Purpose: Play or save the audio\n")
|
||||
overview_parts.append("```\n\n")
|
||||
|
||||
if has_transcript:
|
||||
overview_parts.append("### Get Timestamped Transcript\n")
|
||||
overview_parts.append("### Get Timestamps Transcript\n")
|
||||
overview_parts.append("```python\n")
|
||||
overview_parts.append("timestamps = await audio_resource.timestamps()\n")
|
||||
overview_parts.append("# Returns: FileContent object or None\n")
|
||||
overview_parts.append("# Purpose: Extract timestamped transcript from the audio\n")
|
||||
overview_parts.append("```\n\n")
|
||||
|
||||
overview_parts.append("### Get Audio Metadata\n")
|
||||
@@ -505,22 +245,17 @@ class AudioParser(BaseParser):
|
||||
overview_parts.append(
|
||||
f"channels = audio_resource.get_channels() # {node.meta['channels']}\n"
|
||||
)
|
||||
overview_parts.append(
|
||||
f'format = audio_resource.get_format() # "{node.meta["format"]}"\n'
|
||||
)
|
||||
overview_parts.append(f'format = audio_resource.get_format() # "{node.meta["format"]}"\n')
|
||||
overview_parts.append("```\n")
|
||||
|
||||
overview = "".join(overview_parts)
|
||||
|
||||
# Store in node meta
|
||||
node.meta["abstract"] = abstract
|
||||
node.meta["overview"] = overview
|
||||
|
||||
async def parse_content(
|
||||
self,
|
||||
content: str,
|
||||
source_path: Optional[str] = None,
|
||||
instruction: str = "",
|
||||
**kwargs,
|
||||
self, content: str, source_path: Optional[str] = None, instruction: str = "", **kwargs
|
||||
) -> ParseResult:
|
||||
"""
|
||||
Parse audio from content string - Not yet implemented.
|
||||
@@ -528,7 +263,6 @@ class AudioParser(BaseParser):
|
||||
Args:
|
||||
content: Audio content (base64 or binary string)
|
||||
source_path: Optional source path for metadata
|
||||
instruction: Processing instruction
|
||||
**kwargs: Additional parsing parameters
|
||||
|
||||
Returns:
|
||||
|
||||
@@ -1,44 +0,0 @@
|
||||
metadata:
|
||||
id: "parsing.audio_summary"
|
||||
name: "Audio Summary"
|
||||
description: "Generate concise audio summary from transcript for semantic parsing"
|
||||
version: "1.0.0"
|
||||
language: "en"
|
||||
category: "parsing"
|
||||
|
||||
variables:
|
||||
- name: "transcript"
|
||||
type: "string"
|
||||
description: "Full audio transcript text"
|
||||
required: true
|
||||
max_length: 30000
|
||||
- name: "duration"
|
||||
type: "string"
|
||||
description: "Audio duration in seconds"
|
||||
default: "unknown"
|
||||
required: false
|
||||
- name: "format"
|
||||
type: "string"
|
||||
description: "Audio file format"
|
||||
default: "unknown"
|
||||
required: false
|
||||
|
||||
template: |
|
||||
Please analyze this audio transcript and generate a concise summary for semantic indexing.
|
||||
|
||||
Audio duration: {{ duration }}s
|
||||
Audio format: {{ format }}
|
||||
|
||||
Transcript:
|
||||
{{ transcript }}
|
||||
|
||||
Generate a comprehensive summary that includes:
|
||||
1. Main topic or subject of the audio
|
||||
2. Key points discussed
|
||||
3. Any notable speakers or perspectives
|
||||
4. Important conclusions or takeaways
|
||||
|
||||
Keep the summary clear and factual, suitable for semantic search and understanding.
|
||||
|
||||
llm_config:
|
||||
temperature: 0.0
|
||||
@@ -1,288 +0,0 @@
|
||||
# Copyright (c) 2026 Beijing Volcano Engine Technology Co., Ltd.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Unit tests for AudioParser with mocked Whisper API and mutagen."""
|
||||
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from openviking.parse.base import NodeType
|
||||
from openviking.parse.parsers.media.audio import (
|
||||
AUDIO_MAGIC_BYTES,
|
||||
AudioParser,
|
||||
_extract_metadata_mutagen,
|
||||
_format_timestamp,
|
||||
)
|
||||
from openviking_cli.utils.config.parser_config import AudioConfig
|
||||
|
||||
|
||||
class TestFormatTimestamp:
|
||||
def test_seconds_only(self):
|
||||
assert _format_timestamp(45) == "0:45"
|
||||
|
||||
def test_minutes_and_seconds(self):
|
||||
assert _format_timestamp(125) == "2:05"
|
||||
|
||||
def test_hours(self):
|
||||
assert _format_timestamp(3661) == "1:01:01"
|
||||
|
||||
def test_zero(self):
|
||||
assert _format_timestamp(0) == "0:00"
|
||||
|
||||
|
||||
class TestExtractMetadataMutagen:
|
||||
@patch("openviking.parse.parsers.media.audio._try_import_mutagen")
|
||||
def test_mutagen_not_installed(self, mock_import):
|
||||
mock_import.return_value = None
|
||||
result = _extract_metadata_mutagen(Path("/fake/audio.mp3"))
|
||||
assert result == {}
|
||||
|
||||
@patch("openviking.parse.parsers.media.audio._try_import_mutagen")
|
||||
def test_mutagen_returns_metadata(self, mock_import):
|
||||
mock_mutagen = MagicMock()
|
||||
mock_audio = MagicMock()
|
||||
mock_audio.info.length = 120.5
|
||||
mock_audio.info.sample_rate = 44100
|
||||
mock_audio.info.channels = 2
|
||||
mock_audio.info.bitrate = 320000
|
||||
mock_mutagen.File.return_value = mock_audio
|
||||
mock_import.return_value = mock_mutagen
|
||||
|
||||
result = _extract_metadata_mutagen(Path("/fake/audio.mp3"))
|
||||
assert result["duration"] == 120.5
|
||||
assert result["sample_rate"] == 44100
|
||||
assert result["channels"] == 2
|
||||
assert result["bitrate"] == 320000
|
||||
|
||||
@patch("openviking.parse.parsers.media.audio._try_import_mutagen")
|
||||
def test_mutagen_file_returns_none(self, mock_import):
|
||||
mock_mutagen = MagicMock()
|
||||
mock_mutagen.File.return_value = None
|
||||
mock_import.return_value = mock_mutagen
|
||||
|
||||
result = _extract_metadata_mutagen(Path("/fake/audio.mp3"))
|
||||
assert result == {}
|
||||
|
||||
@patch("openviking.parse.parsers.media.audio._try_import_mutagen")
|
||||
def test_mutagen_raises_exception(self, mock_import):
|
||||
mock_mutagen = MagicMock()
|
||||
mock_mutagen.File.side_effect = Exception("corrupt file")
|
||||
mock_import.return_value = mock_mutagen
|
||||
|
||||
result = _extract_metadata_mutagen(Path("/fake/audio.mp3"))
|
||||
assert result == {}
|
||||
|
||||
|
||||
class TestAudioParserInit:
|
||||
def test_default_config(self):
|
||||
parser = AudioParser()
|
||||
assert parser.config.enable_transcription is True
|
||||
assert parser.config.transcription_model == "whisper-large-v3"
|
||||
|
||||
def test_custom_config(self):
|
||||
config = AudioConfig(enable_transcription=False, language="en")
|
||||
parser = AudioParser(config=config)
|
||||
assert parser.config.enable_transcription is False
|
||||
assert parser.config.language == "en"
|
||||
|
||||
def test_supported_extensions(self):
|
||||
parser = AudioParser()
|
||||
exts = parser.supported_extensions
|
||||
assert ".mp3" in exts
|
||||
assert ".wav" in exts
|
||||
assert ".ogg" in exts
|
||||
assert ".flac" in exts
|
||||
assert ".aac" in exts
|
||||
assert ".m4a" in exts
|
||||
|
||||
def test_can_parse(self):
|
||||
parser = AudioParser()
|
||||
assert parser.can_parse("test.mp3") is True
|
||||
assert parser.can_parse("test.wav") is True
|
||||
assert parser.can_parse("test.txt") is False
|
||||
assert parser.can_parse("test.pdf") is False
|
||||
|
||||
|
||||
class TestAudioParserValidation:
|
||||
def test_validate_mp3_id3(self):
|
||||
parser = AudioParser()
|
||||
audio_bytes = b"ID3" + b"\x00" * 100
|
||||
parser._validate_audio_bytes(audio_bytes, ".mp3", Path("test.mp3"))
|
||||
|
||||
def test_validate_wav_riff(self):
|
||||
parser = AudioParser()
|
||||
audio_bytes = b"RIFF" + b"\x00" * 100
|
||||
parser._validate_audio_bytes(audio_bytes, ".wav", Path("test.wav"))
|
||||
|
||||
def test_validate_flac(self):
|
||||
parser = AudioParser()
|
||||
audio_bytes = b"fLaC" + b"\x00" * 100
|
||||
parser._validate_audio_bytes(audio_bytes, ".flac", Path("test.flac"))
|
||||
|
||||
def test_validate_ogg(self):
|
||||
parser = AudioParser()
|
||||
audio_bytes = b"OggS" + b"\x00" * 100
|
||||
parser._validate_audio_bytes(audio_bytes, ".ogg", Path("test.ogg"))
|
||||
|
||||
def test_invalid_mp3_raises(self):
|
||||
parser = AudioParser()
|
||||
audio_bytes = b"NOT_MP3" + b"\x00" * 100
|
||||
with pytest.raises(ValueError, match="Invalid audio file"):
|
||||
parser._validate_audio_bytes(audio_bytes, ".mp3", Path("test.mp3"))
|
||||
|
||||
def test_unknown_extension_skips_validation(self):
|
||||
parser = AudioParser()
|
||||
audio_bytes = b"anything"
|
||||
parser._validate_audio_bytes(audio_bytes, ".xyz", Path("test.xyz"))
|
||||
|
||||
|
||||
class TestAudioParserParse:
|
||||
@pytest.mark.asyncio
|
||||
async def test_file_not_found(self):
|
||||
parser = AudioParser()
|
||||
with pytest.raises(FileNotFoundError, match="Audio file not found"):
|
||||
await parser.parse("/nonexistent/audio.mp3")
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@patch("openviking.parse.parsers.media.audio._extract_metadata_mutagen")
|
||||
async def test_parse_metadata_only(self, mock_metadata):
|
||||
"""Test parsing with transcription disabled - metadata only."""
|
||||
mock_metadata.return_value = {
|
||||
"duration": 60.0,
|
||||
"sample_rate": 44100,
|
||||
"channels": 2,
|
||||
"bitrate": 128000,
|
||||
}
|
||||
|
||||
config = AudioConfig(enable_transcription=False)
|
||||
parser = AudioParser(config=config)
|
||||
|
||||
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as f:
|
||||
f.write(b"ID3" + b"\x00" * 200)
|
||||
tmp_path = f.name
|
||||
|
||||
try:
|
||||
mock_viking_fs = MagicMock()
|
||||
mock_viking_fs.create_temp_uri.return_value = "viking://temp/test123"
|
||||
mock_viking_fs.mkdir = AsyncMock()
|
||||
mock_viking_fs.write_file_bytes = AsyncMock()
|
||||
mock_viking_fs.write_file = AsyncMock()
|
||||
|
||||
with patch(
|
||||
"openviking.parse.parsers.media.audio.get_viking_fs",
|
||||
return_value=mock_viking_fs,
|
||||
):
|
||||
result = await parser.parse(tmp_path)
|
||||
|
||||
assert result.parser_name == "AudioParser"
|
||||
assert result.source_format == "audio"
|
||||
assert result.root.type == NodeType.ROOT
|
||||
assert result.root.meta["duration"] == 60.0
|
||||
assert result.root.meta["sample_rate"] == 44100
|
||||
assert result.root.meta["channels"] == 2
|
||||
assert result.root.meta["has_transcript"] is False
|
||||
assert len(result.warnings) > 0
|
||||
finally:
|
||||
Path(tmp_path).unlink(missing_ok=True)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@patch("openviking.parse.parsers.media.audio._extract_metadata_mutagen")
|
||||
async def test_parse_with_transcript_segments(self, mock_metadata):
|
||||
"""Test parsing with mocked Whisper returning timestamped segments."""
|
||||
mock_metadata.return_value = {
|
||||
"duration": 30.0,
|
||||
"sample_rate": 16000,
|
||||
"channels": 1,
|
||||
"bitrate": 64000,
|
||||
}
|
||||
|
||||
config = AudioConfig(enable_transcription=True)
|
||||
parser = AudioParser(config=config)
|
||||
|
||||
segments = [
|
||||
{"start": 0.0, "end": 10.0, "text": "Hello world."},
|
||||
{"start": 10.0, "end": 20.0, "text": "This is a test."},
|
||||
{"start": 20.0, "end": 30.0, "text": "Goodbye."},
|
||||
]
|
||||
|
||||
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as f:
|
||||
f.write(b"ID3" + b"\x00" * 200)
|
||||
tmp_path = f.name
|
||||
|
||||
try:
|
||||
mock_viking_fs = MagicMock()
|
||||
mock_viking_fs.create_temp_uri.return_value = "viking://temp/test456"
|
||||
mock_viking_fs.mkdir = AsyncMock()
|
||||
mock_viking_fs.write_file_bytes = AsyncMock()
|
||||
mock_viking_fs.write_file = AsyncMock()
|
||||
|
||||
with (
|
||||
patch(
|
||||
"openviking.parse.parsers.media.audio.get_viking_fs",
|
||||
return_value=mock_viking_fs,
|
||||
),
|
||||
patch.object(
|
||||
parser,
|
||||
"_asr_transcribe_with_timestamps",
|
||||
new_callable=AsyncMock,
|
||||
return_value=segments,
|
||||
),
|
||||
):
|
||||
result = await parser.parse(tmp_path)
|
||||
|
||||
assert result.root.meta["has_transcript"] is True
|
||||
assert result.root.meta["segment_count"] == 3
|
||||
assert len(result.root.children) == 3
|
||||
assert result.root.children[0].type == NodeType.SECTION
|
||||
assert "0:00" in result.root.children[0].title
|
||||
assert result.root.children[0].meta["text"] == "Hello world."
|
||||
assert len(result.warnings) == 0
|
||||
|
||||
mock_viking_fs.write_file.assert_called_once()
|
||||
call_args = mock_viking_fs.write_file.call_args
|
||||
assert "transcript.md" in call_args[0][0]
|
||||
finally:
|
||||
Path(tmp_path).unlink(missing_ok=True)
|
||||
|
||||
|
||||
class TestAudioParserTranscript:
|
||||
def test_build_transcript_markdown_with_segments(self):
|
||||
parser = AudioParser()
|
||||
segments = [
|
||||
{"start": 0.0, "end": 15.0, "text": "First segment."},
|
||||
{"start": 15.0, "end": 30.0, "text": "Second segment."},
|
||||
]
|
||||
md = parser._build_transcript_markdown(segments, "", "test_audio")
|
||||
assert "# Transcript: test_audio" in md
|
||||
assert "**[0:00 - 0:15]** First segment." in md
|
||||
assert "**[0:15 - 0:30]** Second segment." in md
|
||||
|
||||
def test_build_transcript_markdown_plain(self):
|
||||
parser = AudioParser()
|
||||
md = parser._build_transcript_markdown(
|
||||
[], "This is the full transcript text.", "test_audio"
|
||||
)
|
||||
assert "# Transcript: test_audio" in md
|
||||
assert "This is the full transcript text." in md
|
||||
|
||||
|
||||
class TestAudioParserParseContent:
|
||||
@pytest.mark.asyncio
|
||||
async def test_parse_content_not_implemented(self):
|
||||
parser = AudioParser()
|
||||
with pytest.raises(NotImplementedError):
|
||||
await parser.parse_content("base64data")
|
||||
|
||||
|
||||
class TestAudioMagicBytes:
|
||||
def test_magic_bytes_defined(self):
|
||||
"""Verify magic bytes are defined for all supported formats."""
|
||||
assert ".mp3" in AUDIO_MAGIC_BYTES
|
||||
assert ".wav" in AUDIO_MAGIC_BYTES
|
||||
assert ".ogg" in AUDIO_MAGIC_BYTES
|
||||
assert ".flac" in AUDIO_MAGIC_BYTES
|
||||
assert ".aac" in AUDIO_MAGIC_BYTES
|
||||
assert ".m4a" in AUDIO_MAGIC_BYTES
|
||||
assert ".opus" in AUDIO_MAGIC_BYTES
|
||||
Reference in New Issue
Block a user