Files
NSCT---Neutral-Search-Crawl…/tests/models/test_audio.py

637 lines
23 KiB
Python

"""Tests für Pydantic v2 Schemas und SQLAlchemy Models der Audio Evidence Extraction (Stage 11)."""
from __future__ import annotations
import pytest
from pydantic import ValidationError
from nsct.models.audio import (
AudioClaimSchema,
AudioRequestSchema,
AudioReportSchema,
AudioSegmentType,
AudioSpeakerType,
AudioTranscriptSegmentSchema,
)
# ---------------------------------------------------------------------------
# Enums
# ---------------------------------------------------------------------------
class TestAudioSegmentTypeEnum:
"""Prüft die Enums AudioSegmentType und AudioSpeakerType."""
def test_audio_segment_type_values(self):
assert AudioSegmentType.INTERVIEW.value == "interview"
assert AudioSegmentType.PODCAST.value == "podcast"
assert AudioSegmentType.PRESSEKONFERENZ.value == "pressekonferenz"
assert AudioSegmentType.REDEN.value == "reden"
assert AudioSegmentType.SONSTIGE.value == "sonstige"
def test_audio_speaker_type_values(self):
assert AudioSpeakerType.SPOECHTENANTWORTER.value == "sprechantenworter"
assert AudioSpeakerType.FRAGENSTELLER.value == "fragensteller"
assert AudioSpeakerType.MODERATOR.value == "moderator"
assert AudioSpeakerType.SONSTIGE.value == "sonstige"
def test_invalid_segment_type(self):
with pytest.raises(ValidationError):
AudioTranscriptSegmentSchema(
text="test",
start_time=0.0,
end_time=1.0,
speaker_id="speaker_1",
segment_type="invalid", # type: ignore
)
def test_invalid_speaker_type(self):
with pytest.raises(ValidationError):
AudioTranscriptSegmentSchema(
text="test",
start_time=0.0,
end_time=1.0,
speaker_id="speaker_1",
speaker_type="invalid", # type: ignore
)
def test_all_segment_types(self):
for st in AudioSegmentType:
schema = AudioTranscriptSegmentSchema(
text="test content",
start_time=0.0,
end_time=1.0,
speaker_id="speaker_1",
)
assert isinstance(schema, AudioTranscriptSegmentSchema)
def test_all_speaker_types(self):
for st in AudioSpeakerType:
schema = AudioTranscriptSegmentSchema(
text="test content",
start_time=0.0,
end_time=1.0,
speaker_id="speaker_1",
)
assert isinstance(schema, AudioTranscriptSegmentSchema)
# ---------------------------------------------------------------------------
# AudioTranscriptSegmentSchema
# ---------------------------------------------------------------------------
class TestAudioTranscriptSegmentSchema:
"""Tests für AudioTranscriptSegmentSchema — Pflichtfelder, Defaults, Frozen."""
@pytest.fixture
def base_kwargs(self):
return {
"text": "Dies ist ein Testtranskript.",
"start_time": 0.0,
"end_time": 5.0,
"speaker_id": "speaker_1",
}
def test_create_valid_segment(self, base_kwargs):
segment = AudioTranscriptSegmentSchema(**base_kwargs)
assert segment.text == "Dies ist ein Testtranskript."
assert segment.start_time == 0.0
assert segment.end_time == 5.0
assert segment.speaker_id == "speaker_1"
assert segment.confidence == 0.5
def test_defaults(self, base_kwargs):
segment = AudioTranscriptSegmentSchema(**base_kwargs)
assert segment.confidence == 0.5
def test_frozen(self, base_kwargs):
segment = AudioTranscriptSegmentSchema(**base_kwargs)
with pytest.raises(Exception):
segment.text = "modified"
def test_missing_text(self, base_kwargs):
with pytest.raises(ValidationError):
AudioTranscriptSegmentSchema(
start_time=base_kwargs["start_time"],
end_time=base_kwargs["end_time"],
speaker_id=base_kwargs["speaker_id"],
)
def test_empty_text(self, base_kwargs):
with pytest.raises(ValidationError, match="text darf nicht nur aus Whitespaces bestehen"):
AudioTranscriptSegmentSchema(
text=" ",
start_time=base_kwargs["start_time"],
end_time=base_kwargs["end_time"],
speaker_id=base_kwargs["speaker_id"],
)
def test_missing_speaker_id(self, base_kwargs):
with pytest.raises(ValidationError):
AudioTranscriptSegmentSchema(
text=base_kwargs["text"],
start_time=base_kwargs["start_time"],
end_time=base_kwargs["end_time"],
)
def test_empty_speaker_id(self, base_kwargs):
with pytest.raises(ValidationError, match="speaker_id darf nicht leer sein"):
AudioTranscriptSegmentSchema(
text=base_kwargs["text"],
start_time=base_kwargs["start_time"],
end_time=base_kwargs["end_time"],
speaker_id=" ",
)
def test_missing_start_time(self, base_kwargs):
with pytest.raises(ValidationError):
AudioTranscriptSegmentSchema(
text=base_kwargs["text"],
end_time=base_kwargs["end_time"],
speaker_id=base_kwargs["speaker_id"],
)
def test_missing_end_time(self, base_kwargs):
with pytest.raises(ValidationError):
AudioTranscriptSegmentSchema(
text=base_kwargs["text"],
start_time=base_kwargs["start_time"],
speaker_id=base_kwargs["speaker_id"],
)
def test_negative_start_time(self, base_kwargs):
with pytest.raises(ValidationError):
AudioTranscriptSegmentSchema(
text=base_kwargs["text"],
start_time=-1.0,
end_time=base_kwargs["end_time"],
speaker_id=base_kwargs["speaker_id"],
)
def test_end_before_start(self):
with pytest.raises(ValidationError, match="end_time muss nach start_time liegen"):
AudioTranscriptSegmentSchema(
text="test",
start_time=10.0,
end_time=5.0,
speaker_id="speaker_1",
)
def test_confidence_bounds(self, base_kwargs):
schema_low = AudioTranscriptSegmentSchema(**base_kwargs, confidence=0.0)
assert schema_low.confidence == 0.0
schema_high = AudioTranscriptSegmentSchema(**base_kwargs, confidence=1.0)
assert schema_high.confidence == 1.0
def test_confidence_out_of_bounds_low(self, base_kwargs):
with pytest.raises(ValidationError):
AudioTranscriptSegmentSchema(**base_kwargs, confidence=-0.1)
def test_confidence_out_of_bounds_high(self, base_kwargs):
with pytest.raises(ValidationError):
AudioTranscriptSegmentSchema(**base_kwargs, confidence=1.1)
def test_high_confidence(self, base_kwargs):
segment = AudioTranscriptSegmentSchema(**base_kwargs, confidence=0.95)
assert segment.confidence == 0.95
# ---------------------------------------------------------------------------
# AudioClaimSchema
# ---------------------------------------------------------------------------
class TestAudioClaimSchema:
"""Tests für AudioClaimSchema — Claim mit Provenance und Timestamp."""
@pytest.fixture
def base_kwargs(self):
return {
"claim_text": "Die Regierung hat die Ausgaben erhöht.",
"timestamp_start": 120.0,
"timestamp_end": 125.0,
"speaker_id": "minister_1",
"source_url": "https://example.com/interview.mp3",
}
def test_create_valid_claim(self, base_kwargs):
claim = AudioClaimSchema(**base_kwargs)
assert claim.claim_text == "Die Regierung hat die Ausgaben erhöht."
assert claim.timestamp_start == 120.0
assert claim.timestamp_end == 125.0
assert claim.speaker_id == "minister_1"
assert claim.source_url == "https://example.com/interview.mp3"
assert claim.confidence == 0.5
assert claim.evidence_span is None
assert claim.claim_type is None
def test_defaults(self, base_kwargs):
claim = AudioClaimSchema(**base_kwargs)
assert claim.confidence == 0.5
assert claim.evidence_span is None
assert claim.claim_type is None
def test_frozen(self, base_kwargs):
claim = AudioClaimSchema(**base_kwargs)
with pytest.raises(Exception):
claim.claim_text = "modified"
def test_with_evidence_span(self, base_kwargs):
claim = AudioClaimSchema(
**base_kwargs,
evidence_span="Laut dem Haushaltsgesetz 2024 wurden die Ausgaben um 15% erhöht.",
)
assert claim.evidence_span == "Laut dem Haushaltsgesetz 2024 wurden die Ausgaben um 15% erhöht."
def test_with_claim_type(self, base_kwargs):
claim = AudioClaimSchema(**base_kwargs, claim_type="factual")
assert claim.claim_type == "factual"
def test_empty_claim_text(self, base_kwargs):
with pytest.raises(ValidationError, match="claim_text darf nicht nur aus Whitespaces bestehen"):
AudioClaimSchema(
claim_text=" ",
timestamp_start=base_kwargs["timestamp_start"],
timestamp_end=base_kwargs["timestamp_end"],
speaker_id=base_kwargs["speaker_id"],
source_url=base_kwargs["source_url"],
)
def test_missing_claim_text(self, base_kwargs):
with pytest.raises(ValidationError):
AudioClaimSchema(
timestamp_start=base_kwargs["timestamp_start"],
timestamp_end=base_kwargs["timestamp_end"],
speaker_id=base_kwargs["speaker_id"],
source_url=base_kwargs["source_url"],
)
def test_missing_speaker_id(self, base_kwargs):
with pytest.raises(ValidationError):
AudioClaimSchema(
claim_text=base_kwargs["claim_text"],
timestamp_start=base_kwargs["timestamp_start"],
timestamp_end=base_kwargs["timestamp_end"],
source_url=base_kwargs["source_url"],
)
def test_empty_speaker_id(self, base_kwargs):
with pytest.raises(ValidationError, match="speaker_id darf nicht leer sein"):
AudioClaimSchema(
claim_text=base_kwargs["claim_text"],
timestamp_start=base_kwargs["timestamp_start"],
timestamp_end=base_kwargs["timestamp_end"],
speaker_id=" ",
source_url=base_kwargs["source_url"],
)
def test_missing_source_url(self, base_kwargs):
with pytest.raises(ValidationError):
AudioClaimSchema(
claim_text=base_kwargs["claim_text"],
timestamp_start=base_kwargs["timestamp_start"],
timestamp_end=base_kwargs["timestamp_end"],
speaker_id=base_kwargs["speaker_id"],
)
def test_empty_source_url(self, base_kwargs):
with pytest.raises(ValidationError, match="source_url darf nicht leer sein"):
AudioClaimSchema(
claim_text=base_kwargs["claim_text"],
timestamp_start=base_kwargs["timestamp_start"],
timestamp_end=base_kwargs["timestamp_end"],
speaker_id=base_kwargs["speaker_id"],
source_url=" ",
)
def test_timestamp_end_before_start(self):
with pytest.raises(ValidationError, match="timestamp_end muss nach timestamp_start liegen"):
AudioClaimSchema(
claim_text="test",
timestamp_start=10.0,
timestamp_end=5.0,
speaker_id="speaker_1",
source_url="https://example.com",
)
def test_timestamp_bounds(self):
claim = AudioClaimSchema(
claim_text="test",
timestamp_start=0.0,
timestamp_end=0.0,
speaker_id="speaker_1",
source_url="https://example.com",
)
assert claim.timestamp_start == 0.0
assert claim.timestamp_end == 0.0
def test_negative_timestamp_start(self):
with pytest.raises(ValidationError):
AudioClaimSchema(
claim_text="test",
timestamp_start=-1.0,
timestamp_end=10.0,
speaker_id="speaker_1",
source_url="https://example.com",
)
def test_confidence_bounds(self, base_kwargs):
schema_low = AudioClaimSchema(**base_kwargs, confidence=0.0)
assert schema_low.confidence == 0.0
schema_high = AudioClaimSchema(**base_kwargs, confidence=1.0)
assert schema_high.confidence == 1.0
# ---------------------------------------------------------------------------
# AudioReportSchema
# ---------------------------------------------------------------------------
class TestAudioReportSchema:
"""Tests für AudioReportSchema — Zusammenfassung der Audio-Analyse."""
@pytest.fixture
def base_kwargs(self):
return {
"duration_seconds": 3600.0,
"language": "de",
}
def test_create_valid_report(self, base_kwargs):
report = AudioReportSchema(**base_kwargs)
assert report.duration_seconds == 3600.0
assert report.language == "de"
assert report.transcript_segments == []
assert report.claims == []
assert report.source_url is None
assert report.research_run_id is None
assert report.metadata == {}
def test_defaults(self, base_kwargs):
report = AudioReportSchema(**base_kwargs)
assert report.transcript_segments == []
assert report.claims == []
assert report.source_url is None
assert report.research_run_id is None
assert report.metadata == {}
def test_frozen(self, base_kwargs):
report = AudioReportSchema(**base_kwargs)
with pytest.raises(Exception):
report.duration_seconds = 7200.0
def test_with_transcript_segments(self, base_kwargs):
segment = AudioTranscriptSegmentSchema(
text="Guten Tag, ich möchte Sie etwas fragen.",
start_time=0.0,
end_time=3.0,
speaker_id="interviewer",
)
report = AudioReportSchema(
**base_kwargs,
transcript_segments=[segment],
)
assert len(report.transcript_segments) == 1
assert report.transcript_segments[0].text == "Guten Tag, ich möchte Sie etwas fragen."
def test_with_claims(self, base_kwargs):
claim = AudioClaimSchema(
claim_text="Die Regierung hat die Ausgaben erhöht.",
timestamp_start=120.0,
timestamp_end=125.0,
speaker_id="minister_1",
source_url="https://example.com/interview.mp3",
)
report = AudioReportSchema(
**base_kwargs,
claims=[claim],
)
assert len(report.claims) == 1
assert report.claims[0].claim_text == "Die Regierung hat die Ausgaben erhöht."
def test_with_source_url(self, base_kwargs):
report = AudioReportSchema(
**base_kwargs,
source_url="https://example.com/podcast.mp3",
)
assert report.source_url == "https://example.com/podcast.mp3"
def test_with_research_run_id(self, base_kwargs):
report = AudioReportSchema(
**base_kwargs,
research_run_id="run-uuid-001",
)
assert report.research_run_id == "run-uuid-001"
def test_empty_source_url(self, base_kwargs):
with pytest.raises(ValidationError, match="source_url darf nicht leer sein"):
AudioReportSchema(
**base_kwargs,
source_url=" ",
)
def test_empty_language(self, base_kwargs):
with pytest.raises(ValidationError, match="language darf nicht leer sein"):
AudioReportSchema(
**base_kwargs,
language=" ",
)
def test_language_normalized_to_lower(self, base_kwargs):
report = AudioReportSchema(
**base_kwargs,
language="DE",
)
assert report.language == "de"
def test_negative_duration(self):
with pytest.raises(ValidationError):
AudioReportSchema(
duration_seconds=-1.0,
language="de",
)
def test_zero_duration(self):
report = AudioReportSchema(
duration_seconds=0.0,
language="de",
)
assert report.duration_seconds == 0.0
def test_metadata_dict(self, base_kwargs):
report = AudioReportSchema(
**base_kwargs,
metadata={"model": "whisper-3", "processing_time": 12.5},
)
assert report.metadata["model"] == "whisper-3"
assert report.metadata["processing_time"] == 12.5
def test_full_report(self, base_kwargs):
segment = AudioTranscriptSegmentSchema(
text="Interview: Was denken Sie über die Wirtschaftslage?",
start_time=0.0,
end_time=5.0,
speaker_id="interviewer",
)
claim = AudioClaimSchema(
claim_text="Die Wirtschaftslage ist stabil.",
timestamp_start=5.0,
timestamp_end=10.0,
speaker_id="interviewee",
source_url="https://example.com/interview.mp3",
)
report = AudioReportSchema(
**base_kwargs,
transcript_segments=[segment],
claims=[claim],
source_url="https://example.com/interview.mp3",
research_run_id="run-uuid-001",
metadata={"model": "whisper-3"},
)
assert len(report.transcript_segments) == 1
assert len(report.claims) == 1
assert report.source_url == "https://example.com/interview.mp3"
assert report.research_run_id == "run-uuid-001"
def test_language_short_code(self, base_kwargs):
"""Kurze ISO 639-1 Codes sind erlaubt (min_length=2)."""
report = AudioReportSchema(**base_kwargs, language="en")
assert report.language == "en"
def test_language_long_code(self, base_kwargs):
"""Längere Codes bis max_length=5 sind erlaubt."""
report = AudioReportSchema(**base_kwargs, language="deu")
assert report.language == "deu"
# ---------------------------------------------------------------------------
# AudioRequestSchema
# ---------------------------------------------------------------------------
class TestAudioRequestSchema:
"""Tests für AudioRequestSchema — API-Request."""
@pytest.fixture
def base_kwargs(self):
return {
"research_run_id": "run-uuid-001",
"audio_file_url": "https://example.com/interview.mp3",
"segment_type": AudioSegmentType.INTERVIEW,
}
def test_create_valid_request(self, base_kwargs):
request = AudioRequestSchema(**base_kwargs)
assert request.research_run_id == "run-uuid-001"
assert request.audio_file_url == "https://example.com/interview.mp3"
assert request.audio_bytes_b64 is None
assert request.segment_type == AudioSegmentType.INTERVIEW
assert request.source_id is None
def test_defaults(self, base_kwargs):
request = AudioRequestSchema(**base_kwargs)
assert request.audio_bytes_b64 is None
assert request.segment_type == AudioSegmentType.INTERVIEW
assert request.source_id is None
def test_frozen(self, base_kwargs):
request = AudioRequestSchema(**base_kwargs)
with pytest.raises(Exception):
request.research_run_id = "new-id"
def test_with_audio_bytes_b64(self, base_kwargs):
request = AudioRequestSchema(
**base_kwargs,
audio_file_url=None,
audio_bytes_b64="base64encodedaudiodata==",
)
assert request.audio_file_url is None
assert request.audio_bytes_b64 == "base64encodedaudiodata=="
def test_with_source_id(self, base_kwargs):
request = AudioRequestSchema(
**base_kwargs,
source_id="source-uuid-001",
)
assert request.source_id == "source-uuid-001"
def test_missing_research_run_id(self):
with pytest.raises(ValidationError):
AudioRequestSchema(
audio_file_url="https://example.com/interview.mp3",
segment_type=AudioSegmentType.INTERVIEW,
)
def test_empty_research_run_id(self):
with pytest.raises(ValidationError):
AudioRequestSchema(
research_run_id=" ",
audio_file_url="https://example.com/interview.mp3",
segment_type=AudioSegmentType.INTERVIEW,
)
def test_empty_audio_file_url(self, base_kwargs):
with pytest.raises(ValidationError, match="audio_file_url darf nicht leer sein"):
AudioRequestSchema(
**base_kwargs,
audio_file_url=" ",
)
def test_empty_audio_bytes_b64(self, base_kwargs):
with pytest.raises(ValidationError, match="audio_bytes_b64 darf nicht leer sein"):
AudioRequestSchema(
**base_kwargs,
audio_file_url=None,
audio_bytes_b64=" ",
)
def test_podcast_segment_type(self, base_kwargs):
request = AudioRequestSchema(
**base_kwargs,
segment_type=AudioSegmentType.PODCAST,
)
assert request.segment_type == AudioSegmentType.PODCAST
def test_press_conference_segment_type(self, base_kwargs):
request = AudioRequestSchema(
**base_kwargs,
segment_type=AudioSegmentType.PRESSEKONFERENZ,
)
assert request.segment_type == AudioSegmentType.PRESSEKONFERENZ
def test_speeches_segment_type(self, base_kwargs):
request = AudioRequestSchema(
**base_kwargs,
segment_type=AudioSegmentType.REDEN,
)
assert request.segment_type == AudioSegmentType.REDEN
def test_other_segment_type(self, base_kwargs):
request = AudioRequestSchema(
**base_kwargs,
segment_type=AudioSegmentType.SONSTIGE,
)
assert request.segment_type == AudioSegmentType.SONSTIGE
def test_segment_type_override(self):
for st in AudioSegmentType:
request = AudioRequestSchema(
research_run_id="run-uuid-001",
audio_file_url="https://example.com/audio.mp3",
segment_type=st,
)
assert request.segment_type == st
def test_no_audio_url_or_bytes(self, base_kwargs):
"""Erlaubt: kein audio_file_url UND kein audio_bytes_b64 (beide optional)."""
request = AudioRequestSchema(
**base_kwargs,
audio_file_url=None,
audio_bytes_b64=None,
)
assert request.audio_file_url is None
assert request.audio_bytes_b64 is None