feat(stage8): evidence scoring — transparent multidimensional scores (independence, proximity, support, contradiction, directness, date)
This commit is contained in:
@@ -21,6 +21,7 @@ dependencies = [
|
|||||||
"trafilatura>=2.0.0",
|
"trafilatura>=2.0.0",
|
||||||
"uvicorn>=0.30.0",
|
"uvicorn>=0.30.0",
|
||||||
"structlog>=24.0.0",
|
"structlog>=24.0.0",
|
||||||
|
"openai>=3.3.1",
|
||||||
]
|
]
|
||||||
|
|
||||||
[project.optional-dependencies]
|
[project.optional-dependencies]
|
||||||
|
|||||||
@@ -95,6 +95,10 @@ def create_app() -> FastAPI:
|
|||||||
from nsct.api.stage7 import router as stage7_router
|
from nsct.api.stage7 import router as stage7_router
|
||||||
app.include_router(stage7_router, tags=["research"])
|
app.include_router(stage7_router, tags=["research"])
|
||||||
|
|
||||||
|
# Mount evidence scoring router (Stage 8)
|
||||||
|
from nsct.api.stage8 import router as stage8_router
|
||||||
|
app.include_router(stage8_router, tags=["research"])
|
||||||
|
|
||||||
return app
|
return app
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
275
src/nsct/api/stage8.py
Normal file
275
src/nsct/api/stage8.py
Normal file
@@ -0,0 +1,275 @@
|
|||||||
|
"""Stage 8 API endpoints — Evidence Scoring.
|
||||||
|
|
||||||
|
Endpunkte:
|
||||||
|
POST /research/{run_id}/score-evidence — Triggers Stage 8
|
||||||
|
GET /research/{run_id}/scores — All evidence scores
|
||||||
|
GET /research/{run_id}/score-summary — Score summary with top claims, contradictions, uncertainties
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from typing import Any
|
||||||
|
from uuid import UUID
|
||||||
|
|
||||||
|
from fastapi import APIRouter, HTTPException
|
||||||
|
from pydantic import BaseModel, Field
|
||||||
|
|
||||||
|
from nsct.config import AppSettings
|
||||||
|
from nsct.models.claim import Claim as ClaimModel, ClaimType
|
||||||
|
from nsct.providers.llm import get_provider
|
||||||
|
from nsct.providers.metrics import ProviderMetrics
|
||||||
|
from nsct.stages.stage8_evidence_scoring import Stage8EvidenceScoring
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
router = APIRouter()
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Request / Response Schemas
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
class ScoreEvidenceRequest(BaseModel):
|
||||||
|
"""Request zum Triggern von Stage 8 (Evidence Scoring)."""
|
||||||
|
|
||||||
|
research_run_id: str = Field(
|
||||||
|
...,
|
||||||
|
description="UUID des Research-Runs, für den Scores berechnet werden sollen.",
|
||||||
|
)
|
||||||
|
source_ids: list[str] | None = Field(
|
||||||
|
default=None,
|
||||||
|
description="Optionale Liste von Source-IDs. Wenn None → alle Sources.",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class ScoreResponse(BaseModel):
|
||||||
|
"""Ein einzelner Evidence-Score für einen Claim."""
|
||||||
|
|
||||||
|
claim_id: str = Field(..., description="UUID des Claims")
|
||||||
|
research_run_id: str = Field(..., description="Research-Run-UUID")
|
||||||
|
source_independence_score: float = Field(..., description="Unabhängigkeit der Quelle (0.0-1.0)")
|
||||||
|
primary_source_proximity: float = Field(..., description="Nähe an Primärquelle (0.0-1.0)")
|
||||||
|
cross_source_support: float = Field(..., description="Unterstützung durch andere Quellen (0.0-1.0)")
|
||||||
|
contradiction_level: float = Field(..., description="Widerspruchsniveau: 1.0=keine, 0.0=viele (0.0-1.0)")
|
||||||
|
evidence_directness: float = Field(..., description="Direktheit der Evidenz (0.0-1.0)")
|
||||||
|
date_relevance_score: float = Field(..., description="Aktualität der Quelle (0.0-1.0)")
|
||||||
|
evidence_type: str = Field(..., description="EvidenceType: DIRECT_OBSERVATION|SECONDARY_REPORT|ANALYSIS|OPINION|SPECULATION")
|
||||||
|
raw_scores_json: dict[str, Any] = Field(..., description="Alle Rohdaten für Nachvollziehbarkeit")
|
||||||
|
relation_links: list[dict[str, Any]] = Field(
|
||||||
|
default_factory=list,
|
||||||
|
description="Relations zu anderen Claims",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class ScoreListResponse(BaseModel):
|
||||||
|
"""Alle Evidence-Scores für einen Research-Run."""
|
||||||
|
|
||||||
|
research_run_id: str
|
||||||
|
total_claims: int
|
||||||
|
scored_claims: int
|
||||||
|
scores: list[ScoreResponse]
|
||||||
|
|
||||||
|
|
||||||
|
class ScoreSummaryResponse(BaseModel):
|
||||||
|
"""Zusammenfassung aller Scores."""
|
||||||
|
|
||||||
|
research_run_id: str
|
||||||
|
total_scores: int
|
||||||
|
avg_source_independence: float
|
||||||
|
avg_primary_source_proximity: float
|
||||||
|
avg_cross_source_support: float
|
||||||
|
avg_contradiction_level: float
|
||||||
|
avg_evidence_directness: float
|
||||||
|
avg_date_relevance: float
|
||||||
|
top_support_claims: list[str]
|
||||||
|
top_contradiction_claims: list[str]
|
||||||
|
top_directness_claims: list[str]
|
||||||
|
top_independence_claims: list[str]
|
||||||
|
uncertain_claims: list[str]
|
||||||
|
uncertain_count: int
|
||||||
|
evidence_type_distribution: dict[str, int]
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Helper
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _get_llm_provider():
|
||||||
|
"""Initialisiere den LLM-Provider."""
|
||||||
|
config = AppSettings.from_env()
|
||||||
|
metrics = ProviderMetrics()
|
||||||
|
return get_provider(config, metrics)
|
||||||
|
|
||||||
|
|
||||||
|
def _get_mock_claims(run_id: UUID, source_ids: list[str] | None = None) -> list[ClaimModel]:
|
||||||
|
"""Mock-Daten für Evidence Scoring.
|
||||||
|
|
||||||
|
TODO: In Produktion aus DB laden.
|
||||||
|
"""
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Endpoints
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
@router.post(
|
||||||
|
"/research/{run_id}/score-evidence",
|
||||||
|
response_model=ScoreListResponse,
|
||||||
|
summary="Stage 8 — Trigger Evidence Scoring",
|
||||||
|
)
|
||||||
|
async def score_evidence(
|
||||||
|
run_id: str,
|
||||||
|
request: ScoreEvidenceRequest,
|
||||||
|
) -> ScoreListResponse:
|
||||||
|
"""Startet Stage 8: Evidence Scoring — multidimensionale, transparente Scores.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
run_id : str
|
||||||
|
UUID des Research-Runs.
|
||||||
|
request : ScoreEvidenceRequest
|
||||||
|
Optionale Source-IDs zum Filtern.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
ScoreListResponse
|
||||||
|
Alle Scores mit allen 6 Dimensionen und raw_scores_json.
|
||||||
|
"""
|
||||||
|
if not run_id:
|
||||||
|
raise HTTPException(status_code=400, detail="run_id darf nicht leer sein")
|
||||||
|
|
||||||
|
try:
|
||||||
|
run_uuid = UUID(run_id)
|
||||||
|
except ValueError:
|
||||||
|
raise HTTPException(status_code=400, detail="Ungültige run_id")
|
||||||
|
|
||||||
|
# TODO: In Produktion — Claims aus DB laden
|
||||||
|
mock_claims = _get_mock_claims(run_uuid, request.source_ids)
|
||||||
|
|
||||||
|
if not mock_claims:
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=404,
|
||||||
|
detail=f"Keine Claims für research_run_id={run_id} gefunden",
|
||||||
|
)
|
||||||
|
|
||||||
|
stage8 = Stage8EvidenceScoring(
|
||||||
|
research_run_id=run_uuid,
|
||||||
|
claims=mock_claims,
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
result = stage8.run()
|
||||||
|
except Exception as exc:
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=500,
|
||||||
|
detail=f"Evidence Scoring fehlgeschlagen: {exc}",
|
||||||
|
)
|
||||||
|
|
||||||
|
scores_data = result.get("scores", [])
|
||||||
|
response_scores = [
|
||||||
|
ScoreResponse(
|
||||||
|
claim_id=s["claim_id"],
|
||||||
|
research_run_id=s["research_run_id"],
|
||||||
|
source_independence_score=s["source_independence_score"],
|
||||||
|
primary_source_proximity=s["primary_source_proximity"],
|
||||||
|
cross_source_support=s["cross_source_support"],
|
||||||
|
contradiction_level=s["contradiction_level"],
|
||||||
|
evidence_directness=s["evidence_directness"],
|
||||||
|
date_relevance_score=s["date_relevance_score"],
|
||||||
|
evidence_type=s["evidence_type"],
|
||||||
|
raw_scores_json=s.get("raw_scores_json", {}),
|
||||||
|
relation_links=s.get("relation_links", []),
|
||||||
|
)
|
||||||
|
for s in scores_data
|
||||||
|
]
|
||||||
|
|
||||||
|
return ScoreListResponse(
|
||||||
|
research_run_id=run_id,
|
||||||
|
total_claims=result.get("total_claims", len(scores_data)),
|
||||||
|
scored_claims=result.get("scored_claims", len(scores_data)),
|
||||||
|
scores=response_scores,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get(
|
||||||
|
"/research/{run_id}/scores",
|
||||||
|
response_model=ScoreListResponse,
|
||||||
|
summary="Stage 8 — Liefert alle Evidence-Scores",
|
||||||
|
)
|
||||||
|
async def get_scores(run_id: str) -> ScoreListResponse:
|
||||||
|
"""Liefert alle Evidence-Scores für einen Research-Run.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
run_id : str
|
||||||
|
UUID des Research-Runs.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
ScoreListResponse
|
||||||
|
Alle Scores mit allen 6 Dimensionen.
|
||||||
|
"""
|
||||||
|
if not run_id:
|
||||||
|
raise HTTPException(status_code=400, detail="run_id darf nicht leer sein")
|
||||||
|
|
||||||
|
try:
|
||||||
|
run_uuid = UUID(run_id)
|
||||||
|
except ValueError:
|
||||||
|
raise HTTPException(status_code=400, detail="Ungültige run_id")
|
||||||
|
|
||||||
|
# TODO: In Produktion — Scores aus DB laden
|
||||||
|
return ScoreListResponse(
|
||||||
|
research_run_id=run_id,
|
||||||
|
total_claims=0,
|
||||||
|
scored_claims=0,
|
||||||
|
scores=[],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get(
|
||||||
|
"/research/{run_id}/score-summary",
|
||||||
|
response_model=ScoreSummaryResponse,
|
||||||
|
summary="Stage 8 — Liefert Score-Zusammenfassung",
|
||||||
|
)
|
||||||
|
async def get_score_summary(run_id: str) -> ScoreSummaryResponse:
|
||||||
|
"""Liefert eine Zusammenfassung aller Scores mit Top-Claims, Contradictions, Uncertainties.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
run_id : str
|
||||||
|
UUID des Research-Runs.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
ScoreSummaryResponse
|
||||||
|
Statistiken, Top-Claims, Uncertain-Claims, EvidenceType-Verteilung.
|
||||||
|
"""
|
||||||
|
if not run_id:
|
||||||
|
raise HTTPException(status_code=400, detail="run_id darf nicht leer sein")
|
||||||
|
|
||||||
|
try:
|
||||||
|
run_uuid = UUID(run_id)
|
||||||
|
except ValueError:
|
||||||
|
raise HTTPException(status_code=400, detail="Ungültige run_id")
|
||||||
|
|
||||||
|
# TODO: In Produktion — Scores aus DB laden und Summary berechnen
|
||||||
|
return ScoreSummaryResponse(
|
||||||
|
research_run_id=run_id,
|
||||||
|
total_scores=0,
|
||||||
|
avg_source_independence=0.0,
|
||||||
|
avg_primary_source_proximity=0.0,
|
||||||
|
avg_cross_source_support=0.0,
|
||||||
|
avg_contradiction_level=0.0,
|
||||||
|
avg_evidence_directness=0.0,
|
||||||
|
avg_date_relevance=0.0,
|
||||||
|
top_support_claims=[],
|
||||||
|
top_contradiction_claims=[],
|
||||||
|
top_directness_claims=[],
|
||||||
|
top_independence_claims=[],
|
||||||
|
uncertain_claims=[],
|
||||||
|
uncertain_count=0,
|
||||||
|
evidence_type_distribution={},
|
||||||
|
)
|
||||||
695
src/nsct/stages/stage8_evidence_scoring.py
Normal file
695
src/nsct/stages/stage8_evidence_scoring.py
Normal file
@@ -0,0 +1,695 @@
|
|||||||
|
"""Stage 8: Evidence Scoring — transparente multidimensionale Scores.
|
||||||
|
|
||||||
|
Pipeline für ein Research-Run:
|
||||||
|
1. Lädt alle Claims des Runs aus der DB.
|
||||||
|
2. Für jeden Claim berechnet 6 dimensionale Scores:
|
||||||
|
a) source_independence_score – wie unabhängig ist die Quelle?
|
||||||
|
b) primary_source_proximity – Nähe an einer Primärquelle
|
||||||
|
c) cross_source_support – wie viele Quellen unterstützen denselben Claim?
|
||||||
|
d) contradiction_level – wie stark ist der Dissens?
|
||||||
|
e) evidence_directness – wie direkt ist die Evidenz?
|
||||||
|
f) date_relevance_score – wie aktuell ist die Quelle?
|
||||||
|
3. evidence_type wird klassifiziert (DIRECT_OBSERVATION … SPECULATION).
|
||||||
|
4. Alle Rohwerte werden in raw_scores_json für vollständige Nachvollziehbarkeit gespeichert.
|
||||||
|
5. EvidenceScoreRelationModel verknüpft支持与/oder widersprechende Claims.
|
||||||
|
|
||||||
|
ARCHITEKTUR-REGELN:
|
||||||
|
- Kein einziger \"truth_score\" — mehrere dimensionale Scores
|
||||||
|
- Scores sind transparent — jede Komponente nachvollziehbar
|
||||||
|
- Jede relevante Behauptung benötigt Provenance
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import re
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from typing import Any
|
||||||
|
from uuid import UUID
|
||||||
|
|
||||||
|
from nsct.models.claim import Claim as ClaimModel, ClaimType
|
||||||
|
from nsct.storage.models import (
|
||||||
|
EvidenceScoreModel,
|
||||||
|
EvidenceScoreRelationModel,
|
||||||
|
EvidenceRelationTypeV2,
|
||||||
|
EvidenceType,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Pattern-Mapping für evidence_directness und evidence_type
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
# Wörter/Phrasen die auf direkte Beobachtung hindeuten
|
||||||
|
_DIRECT_PATTERNS: list[str] = [
|
||||||
|
r"zeigen\b", r"belegen\b", r"weisen\s*nach\b", r"ermitteln\b",
|
||||||
|
r"feststellen\b", r"messbar\b", r"quantifiziert\b", r"gemessen\b",
|
||||||
|
r"beobachtet\b", r"verzeichnet\b", r"erhoben\b", r"dokumentiert\b",
|
||||||
|
r"zitiert\b", r"offenbar\b", r"deutlich\s*auf\s*[\w-]+\s*hin",
|
||||||
|
r"ergibt\s*sich", r"führt\s*zu\s*dem\s*Ergebnis",
|
||||||
|
r"die\s*Daten\s*zeigen", r"die\s*Studie\s*zeigt",
|
||||||
|
r"laut\s*Untersuchung", r"nach\s*Angabe\s*von",
|
||||||
|
]
|
||||||
|
|
||||||
|
# Wörter/Phrasen die auf Schlussfolgerung hindeuten
|
||||||
|
_INFERRED_PATTERNS: list[str] = [
|
||||||
|
r"impliziert\b", r"lässt\s*schließen", r"deutet\s*hin\s*auf",
|
||||||
|
r"kann\s*geschlossen\s*werden", r"lässt\s*sich\s*ableiten",
|
||||||
|
r"führt\s*zu\s*dem\s*Schluss", r"folgt\s*daraus",
|
||||||
|
r"erlaubt\s*einen\s*Schluss", r"lässt\s*erwarten",
|
||||||
|
r"weist\s*auf\s*einen\s*Zusammenhang\s*hin",
|
||||||
|
]
|
||||||
|
|
||||||
|
# Wörter/Phrasen die auf Spekulation/Meinung hindeuten
|
||||||
|
_SPECULATION_PATTERNS: list[str] = [
|
||||||
|
r"vielleicht", r"könnte\s*sein", r"dürfte\s*sein",
|
||||||
|
r"vermutet", r"schätzt", r"glaubt", r"vermuten",
|
||||||
|
r"hypothese", r"spekulieren", r"könnte\s*auch",
|
||||||
|
r"wahrscheinlich", r"vielleicht\s*gar", r"evtl\b", r"eventuell",
|
||||||
|
r"potenziell", r"könnte\s*führen", r"würde\s*erlauben",
|
||||||
|
r"scheint\s*so", r"wirkt\s*auf\s*diesen\s*Hinblick",
|
||||||
|
]
|
||||||
|
|
||||||
|
# Wörter/Phrasen die auf Analyse hindeuten
|
||||||
|
_ANALYSIS_PATTERNS: list[str] = [
|
||||||
|
r"analyse\b", r"analysieren", r"betrachtet", r"betrachtung",
|
||||||
|
r"auswertung", r"auswerten", r"bewertung", r"bewerten",
|
||||||
|
r"interpretation", r"deutung", r"einordnung",
|
||||||
|
r"kritisch\s*betrachtet", r"gegenüberstellung",
|
||||||
|
r"vergleichende\s*betrachtung",
|
||||||
|
]
|
||||||
|
|
||||||
|
# Vage Formulierungen
|
||||||
|
_VAGUE_PATTERNS: list[str] = [
|
||||||
|
r"angeblich", r"reportet", r"nach\s*meldungen", r"gerücht",
|
||||||
|
r"anscheinend", r"wohl", r"offenbar", r"angeblich\s*",
|
||||||
|
r"berichten\s*von", r"zitiert\s*werden",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# evidence_type Klassifikation
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _classify_evidence_type(claim_text: str) -> EvidenceType:
|
||||||
|
"""Klassifiziert den Claim-Text in eine EvidenceType-Kategorie.
|
||||||
|
|
||||||
|
Priorität: DIRECT_OBSERVATION > SECONDARY_REPORT > ANALYSIS > OPINION > SPECULATION
|
||||||
|
"""
|
||||||
|
text_lower = claim_text.lower()
|
||||||
|
|
||||||
|
# Check direct observation
|
||||||
|
if any(re.search(p, text_lower) for p in _DIRECT_PATTERNS):
|
||||||
|
return EvidenceType.DIRECT_OBSERVATION
|
||||||
|
|
||||||
|
# Check analysis
|
||||||
|
if any(re.search(p, text_lower) for p in _ANALYSIS_PATTERNS):
|
||||||
|
return EvidenceType.ANALYSIS
|
||||||
|
|
||||||
|
# Check speculation
|
||||||
|
if any(re.search(p, text_lower) for p in _SPECULATION_PATTERNS):
|
||||||
|
return EvidenceType.SPECULATION
|
||||||
|
|
||||||
|
# Check secondary report (reported speech, quotes, references)
|
||||||
|
if any(re.search(p, text_lower) for p in _VAGUE_PATTERNS):
|
||||||
|
return EvidenceType.SECONDARY_REPORT
|
||||||
|
|
||||||
|
# Default: second report / general claim
|
||||||
|
return EvidenceType.SECONDARY_REPORT
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# evidence_directness — Score (0.0–1.0)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _compute_evidence_directness(claim_text: str) -> float:
|
||||||
|
"""Berechnet die Direktheit der Evidenz (0.0–1.0).
|
||||||
|
|
||||||
|
- 1.0 = direkter Befund (Zitat, Beobachtung, spezifische Zahlen)
|
||||||
|
- 0.5 = Schlussfolgerung
|
||||||
|
- 0.0 = Spekulation
|
||||||
|
"""
|
||||||
|
if not claim_text:
|
||||||
|
return 0.5
|
||||||
|
|
||||||
|
text_lower = claim_text.lower()
|
||||||
|
|
||||||
|
# Spekulation → 0.0
|
||||||
|
if any(re.search(p, text_lower) for p in _SPECULATION_PATTERNS):
|
||||||
|
return 0.0
|
||||||
|
|
||||||
|
# Vag/formell → 0.3
|
||||||
|
if any(re.search(p, text_lower) for p in _VAGUE_PATTERNS):
|
||||||
|
return 0.3
|
||||||
|
|
||||||
|
# Analyse → 0.5
|
||||||
|
if any(re.search(p, text_lower) for p in _ANALYSIS_PATTERNS):
|
||||||
|
return 0.5
|
||||||
|
|
||||||
|
# Inference → 0.5
|
||||||
|
if any(re.search(p, text_lower) for p in _INFERRED_PATTERNS):
|
||||||
|
return 0.5
|
||||||
|
|
||||||
|
# Direkte Beobachtung/Zahlen/Zitate → 1.0
|
||||||
|
if any(re.search(p, text_lower) for p in _DIRECT_PATTERNS):
|
||||||
|
return 1.0
|
||||||
|
|
||||||
|
# Enthält spezifische Zahlen → 0.8
|
||||||
|
if re.search(r"\b\d{1,3}(?:\.\d{1,2})?\s*(?:%|°[cCF]|[\s]€/€/€/£\$|[\s]M[\s]|[\s]Mr\.|[\s]\d{3}\b)", text_lower):
|
||||||
|
return 0.8
|
||||||
|
|
||||||
|
# Enthält direkte Zitate → 0.7
|
||||||
|
if re.search(r'["„«].+["„»]', claim_text):
|
||||||
|
return 0.7
|
||||||
|
|
||||||
|
# Default: Schlussfolgerung → 0.5
|
||||||
|
return 0.5
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# date_relevance_score
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _compute_date_relevance_score(publication_date: datetime | None) -> float:
|
||||||
|
"""Score basierend auf der Aktualität der Quelle.
|
||||||
|
|
||||||
|
score = max(0.0, 1.0 - age_days / 365.0) (max 1 Jahr)
|
||||||
|
Wenn kein Datum → 0.5
|
||||||
|
"""
|
||||||
|
if publication_date is None:
|
||||||
|
return 0.5
|
||||||
|
|
||||||
|
now = datetime.now(timezone.utc).replace(tzinfo=None)
|
||||||
|
try:
|
||||||
|
age = now - publication_date
|
||||||
|
except TypeError:
|
||||||
|
# timezone-aware vs naive mismatch
|
||||||
|
age = now - publication_date.replace(tzinfo=None) if publication_date.tzinfo else now - publication_date
|
||||||
|
|
||||||
|
age_days = max(0, age.days)
|
||||||
|
score = max(0.0, 1.0 - age_days / 365.0)
|
||||||
|
return round(score, 4)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# source_independence_score
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _compute_source_independence_score(
|
||||||
|
source_data: dict[str, Any] | None,
|
||||||
|
) -> float:
|
||||||
|
"""Berechnet den independence_score basierend auf Stage-6-Daten.
|
||||||
|
|
||||||
|
- independence_score=1.0 → 1.0
|
||||||
|
- Quelle ist syndiziert → 0.3-0.7 je nach Syndication-Größe
|
||||||
|
- Wenn keine Daten → 0.5 als Default
|
||||||
|
"""
|
||||||
|
if not source_data:
|
||||||
|
return 0.5
|
||||||
|
|
||||||
|
independence = source_data.get("independence_score")
|
||||||
|
if independence is not None:
|
||||||
|
return round(float(independence), 4)
|
||||||
|
|
||||||
|
# Check syndication group
|
||||||
|
synd_group_id = source_data.get("syndication_group_id")
|
||||||
|
if synd_group_id:
|
||||||
|
# Syndiziert: score 0.3-0.7
|
||||||
|
shared_urls = source_data.get("shared_urls", {})
|
||||||
|
shared_count = 0
|
||||||
|
if isinstance(shared_urls, dict):
|
||||||
|
shared_count = len(shared_urls)
|
||||||
|
elif isinstance(shared_urls, list):
|
||||||
|
shared_count = len(shared_urls)
|
||||||
|
# More shared URLs → lower score
|
||||||
|
score = 0.7 - min(0.4, shared_count * 0.1)
|
||||||
|
return round(max(0.3, min(0.7, score)), 4)
|
||||||
|
|
||||||
|
return 0.5
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# primary_source_proximity
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _compute_primary_source_proximity(
|
||||||
|
source_data: dict[str, Any] | None,
|
||||||
|
) -> tuple[float, str]:
|
||||||
|
"""Berechnet primary_source_proximity und gibt (score, label) zurück.
|
||||||
|
|
||||||
|
- DIRECT (1.0): Quelle hat parent_source_id und ist kein Syndicated
|
||||||
|
- SYNDICATED (0.6): Quelle ist in syndication_group
|
||||||
|
- DERIVED (0.3): Keine parent_source_id, nicht syndiziert
|
||||||
|
- UNKNOWN (0.0): Keine Daten
|
||||||
|
"""
|
||||||
|
if not source_data:
|
||||||
|
return 0.0, "UNKNOWN"
|
||||||
|
|
||||||
|
parent_source_id = source_data.get("parent_source_id")
|
||||||
|
synd_group_id = source_data.get("syndication_group_id")
|
||||||
|
independence = source_data.get("independence_score")
|
||||||
|
|
||||||
|
# Hat parent_source_id UND ist nicht selbst syndiziert → DIRECT
|
||||||
|
if parent_source_id and not synd_group_id and (independence is None or independence >= 0.7):
|
||||||
|
return 1.0, "DIRECT"
|
||||||
|
|
||||||
|
# Ist in Syndication-Gruppe → SYNDICATED
|
||||||
|
if synd_group_id:
|
||||||
|
return 0.6, "SYNDICATED"
|
||||||
|
|
||||||
|
# Keine parent_source_id und nicht syndiziert → DERIVED
|
||||||
|
if parent_source_id is None and not synd_group_id:
|
||||||
|
return 0.3, "DERIVED"
|
||||||
|
|
||||||
|
# Fallback → UNKNOWN
|
||||||
|
return 0.0, "UNKNOWN"
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# cross_source_support
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _compute_cross_source_support(
|
||||||
|
claim_id: str,
|
||||||
|
cluster_claims: list[dict[str, Any]],
|
||||||
|
cluster_relations: list[dict[str, Any]],
|
||||||
|
source_independence_map: dict[str, dict[str, Any]],
|
||||||
|
) -> tuple[float, dict[str, Any]]:
|
||||||
|
"""Berechnet cross_source_support basierend auf unterstützenden Quellen.
|
||||||
|
|
||||||
|
Score = (supported_count / total_independent_sources) * weight
|
||||||
|
|
||||||
|
Gibt (score, raw_data_dict) zurück.
|
||||||
|
"""
|
||||||
|
if not cluster_claims:
|
||||||
|
return 0.0, {"supported_count": 0, "total_count": 0, "weight": 1.0}
|
||||||
|
|
||||||
|
# Count unique source_ids in cluster
|
||||||
|
unique_sources: dict[str, dict[str, Any]] = {}
|
||||||
|
for claim_info in cluster_claims:
|
||||||
|
src_id = claim_info.get("source_id", "")
|
||||||
|
if src_id:
|
||||||
|
unique_sources[src_id] = source_independence_map.get(src_id, {
|
||||||
|
"independence_score": 0.5,
|
||||||
|
})
|
||||||
|
|
||||||
|
total_sources = len(unique_sources)
|
||||||
|
if total_sources == 0:
|
||||||
|
return 0.0, {
|
||||||
|
"supported_count": 0,
|
||||||
|
"total_count": 0,
|
||||||
|
"weight": 1.0,
|
||||||
|
}
|
||||||
|
|
||||||
|
# Count SUPPORTS relations for this claim within cluster
|
||||||
|
supports_count = 0
|
||||||
|
total_relations_in_cluster = 0
|
||||||
|
for rel in cluster_relations:
|
||||||
|
total_relations_in_cluster += 1
|
||||||
|
src_id = rel.get("source_claim_id", "")
|
||||||
|
tgt_id = rel.get("target_claim_id", "")
|
||||||
|
rel_type = rel.get("relation_type", "")
|
||||||
|
if rel_type == "SUPPORTS":
|
||||||
|
if src_id == claim_id or tgt_id == claim_id:
|
||||||
|
supports_count += 1
|
||||||
|
|
||||||
|
# If no relations, use direct claim count as fallback
|
||||||
|
if total_relations_in_cluster == 0:
|
||||||
|
score = min(1.0, supports_count / max(1, total_sources))
|
||||||
|
else:
|
||||||
|
score = min(1.0, supports_count / max(1, total_relations_in_cluster))
|
||||||
|
|
||||||
|
return round(score, 4), {
|
||||||
|
"supported_count": supports_count,
|
||||||
|
"total_count": total_sources,
|
||||||
|
"total_relations": total_relations_in_cluster,
|
||||||
|
"weight": 1.0,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# contradiction_level
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _compute_contradiction_level(
|
||||||
|
claim_id: str,
|
||||||
|
cluster_relations: list[dict[str, Any]],
|
||||||
|
) -> tuple[float, dict[str, Any]]:
|
||||||
|
"""Berechnet contradiction_level (1.0 = keine Widersprüche, 0.0 = viele).
|
||||||
|
|
||||||
|
Score = 1.0 - (contradictions / total_relations)
|
||||||
|
"""
|
||||||
|
total_relations = 0
|
||||||
|
contradiction_count = 0
|
||||||
|
|
||||||
|
for rel in cluster_relations:
|
||||||
|
src_id = rel.get("source_claim_id", "")
|
||||||
|
tgt_id = rel.get("target_claim_id", "")
|
||||||
|
rel_type = rel.get("relation_type", "")
|
||||||
|
|
||||||
|
if rel_type in ("CONTRADICTS", "disagrees"):
|
||||||
|
if src_id == claim_id or tgt_id == claim_id:
|
||||||
|
contradiction_count += 1
|
||||||
|
|
||||||
|
total_relations += 1
|
||||||
|
|
||||||
|
if total_relations == 0:
|
||||||
|
return 1.0, {
|
||||||
|
"contradiction_count": 0,
|
||||||
|
"total_relations": 0,
|
||||||
|
"contradiction_ratio": 0.0,
|
||||||
|
}
|
||||||
|
|
||||||
|
score = max(0.0, 1.0 - (contradiction_count / total_relations))
|
||||||
|
ratio = contradiction_count / total_relations
|
||||||
|
|
||||||
|
return round(score, 4), {
|
||||||
|
"contradiction_count": contradiction_count,
|
||||||
|
"total_relations": total_relations,
|
||||||
|
"contradiction_ratio": round(ratio, 4),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Raw-Score Dict
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _build_raw_scores_json(
|
||||||
|
source_independence: float,
|
||||||
|
proximity_label: str,
|
||||||
|
cross_support_data: dict[str, Any],
|
||||||
|
contradiction_data: dict[str, Any],
|
||||||
|
directness_score: float,
|
||||||
|
date_score: float,
|
||||||
|
evidence_type: str,
|
||||||
|
claim_text: str,
|
||||||
|
publication_date: datetime | None,
|
||||||
|
source_data: dict[str, Any] | None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Baut das raw_scores_json mit allen Rohdaten für vollständige Nachvollziehbarkeit."""
|
||||||
|
return {
|
||||||
|
"source_independence": source_independence,
|
||||||
|
"primary_source_proximity": proximity_label,
|
||||||
|
"cross_support_count": cross_support_data.get("supported_count", 0),
|
||||||
|
"cross_total": cross_support_data.get("total_count", 0),
|
||||||
|
"contradiction_count": contradiction_data.get("contradiction_count", 0),
|
||||||
|
"total_relations": contradiction_data.get("total_relations", 0),
|
||||||
|
"contradiction_ratio": contradiction_data.get("contradiction_ratio", 0.0),
|
||||||
|
"evidence_directness": directness_score,
|
||||||
|
"date_relevance_score": date_score,
|
||||||
|
"evidence_type": evidence_type,
|
||||||
|
"claim_text_preview": claim_text[:200] if claim_text else "",
|
||||||
|
"publication_date": publication_date.isoformat() if publication_date else None,
|
||||||
|
"source_has_parent": source_data.get("parent_source_id") if source_data else None,
|
||||||
|
"source_in_syndication_group": source_data.get("syndication_group_id") if source_data else None,
|
||||||
|
"source_independence_raw": source_data.get("independence_score") if source_data else None,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Stage 8 Pipeline
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
class Stage8EvidenceScoring:
|
||||||
|
"""Stage 8: Evidence Scoring — multidimensionale, transparente Scores.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
scoring = Stage8EvidenceScoring(
|
||||||
|
llm_provider=...,
|
||||||
|
config=...,
|
||||||
|
research_run_id=...,
|
||||||
|
claims=claim_models,
|
||||||
|
cluster_data=..., # Stage 7 clusters + relations
|
||||||
|
source_data_map=..., # dict[source_id] -> source independence data
|
||||||
|
)
|
||||||
|
results = scoring.run()
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
research_run_id: UUID,
|
||||||
|
claims: list[ClaimModel],
|
||||||
|
cluster_data: dict[str, Any] | None = None,
|
||||||
|
source_data_map: dict[str, dict[str, Any]] | None = None,
|
||||||
|
):
|
||||||
|
self.research_run_id = research_run_id
|
||||||
|
self.claims = claims
|
||||||
|
self.cluster_data = cluster_data or {"clusters": [], "relations": []}
|
||||||
|
self.source_data_map = source_data_map or {}
|
||||||
|
self._claim_map: dict[str, dict[str, Any]] = {}
|
||||||
|
self._build_claim_map()
|
||||||
|
|
||||||
|
def _build_claim_map(self) -> None:
|
||||||
|
"""Baut eine schnelle Map: claim_id → claim_info."""
|
||||||
|
for claim in self.claims:
|
||||||
|
self._claim_map[str(claim.id)] = {
|
||||||
|
"id": str(claim.id),
|
||||||
|
"text": claim.claim_text or "",
|
||||||
|
"source_id": str(claim.source_id),
|
||||||
|
"research_run_id": str(claim.research_run_id),
|
||||||
|
"claim_type": str(claim.claim_type) if claim.claim_type else "",
|
||||||
|
"confidence": float(claim.confidence) if claim.confidence else 1.0,
|
||||||
|
"evidence_span": claim.evidence_span or "",
|
||||||
|
"event_date": getattr(claim, "event_date", None),
|
||||||
|
}
|
||||||
|
|
||||||
|
def _get_source_data(self, source_id: str) -> dict[str, Any] | None:
|
||||||
|
"""Liefert Source-Independence-Daten für eine Source-ID."""
|
||||||
|
return self.source_data_map.get(source_id)
|
||||||
|
|
||||||
|
def _find_claim_in_clusters(self, claim_id: str) -> tuple[
|
||||||
|
list[dict[str, Any]], # claims in same cluster
|
||||||
|
list[dict[str, Any]], # all relations in cluster
|
||||||
|
]:
|
||||||
|
"""Findet alle Claims und Relations in dem Cluster, in dem sich ein Claim befindet."""
|
||||||
|
cluster_claims: list[dict[str, Any]] = []
|
||||||
|
cluster_relations: list[dict[str, Any]] = []
|
||||||
|
|
||||||
|
# Build relations lookup: cluster_id → list of relations
|
||||||
|
relations_by_cluster: dict[str, list[dict[str, Any]]] = {}
|
||||||
|
for rel in self.cluster_data.get("relations", []):
|
||||||
|
cluster_id = rel.get("cluster_id", "")
|
||||||
|
relations_by_cluster.setdefault(cluster_id, []).append(rel)
|
||||||
|
|
||||||
|
cluster_claims_map: dict[str, list[dict[str, Any]]] = {}
|
||||||
|
for cluster in self.cluster_data.get("clusters", []):
|
||||||
|
cid = cluster.get("id", "") or cluster.get("cluster_id", "")
|
||||||
|
cids = cluster.get("claim_ids", [])
|
||||||
|
if cid:
|
||||||
|
cluster_claims_map[cid] = cids
|
||||||
|
|
||||||
|
for cid, cids in cluster_claims_map.items():
|
||||||
|
if claim_id in cids:
|
||||||
|
# This is the cluster we want
|
||||||
|
for cid_str in cids:
|
||||||
|
if cid_str in self._claim_map:
|
||||||
|
cluster_claims.append(self._claim_map[cid_str])
|
||||||
|
if cid in relations_by_cluster:
|
||||||
|
cluster_relations = list(relations_by_cluster[cid])
|
||||||
|
break
|
||||||
|
|
||||||
|
return cluster_claims, cluster_relations
|
||||||
|
|
||||||
|
def _compute_relation_links(
|
||||||
|
self,
|
||||||
|
claim_id: str,
|
||||||
|
claim_text: str,
|
||||||
|
cluster_relations: list[dict[str, Any]],
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""Berechnet die EvidenceScoreRelationModel-Einträge für einen Claim."""
|
||||||
|
relations = []
|
||||||
|
for rel in cluster_relations:
|
||||||
|
src = rel.get("source_claim_id", "")
|
||||||
|
tgt = rel.get("target_claim_id", "")
|
||||||
|
rel_type_str = rel.get("relation_type", "NEUTRAL")
|
||||||
|
|
||||||
|
# Determine our role in the relation
|
||||||
|
if src == claim_id:
|
||||||
|
our_role = "source"
|
||||||
|
target_id = tgt
|
||||||
|
elif tgt == claim_id:
|
||||||
|
our_role = "target"
|
||||||
|
target_id = src
|
||||||
|
else:
|
||||||
|
continue # not relevant to this claim
|
||||||
|
|
||||||
|
# Map to EvidenceRelationTypeV2
|
||||||
|
if rel_type_str in ("CONTRADICTS", "contradicts", "disagrees"):
|
||||||
|
evidence_rel = EvidenceRelationTypeV2.CONTRADICTS
|
||||||
|
elif rel_type_str in ("SUPPORTS", "supports", "agrees"):
|
||||||
|
evidence_rel = EvidenceRelationTypeV2.SUPPORTS
|
||||||
|
else:
|
||||||
|
evidence_rel = EvidenceRelationTypeV2.NEUTRAL
|
||||||
|
|
||||||
|
# If the relation is about our claim, invert the meaning
|
||||||
|
if our_role == "target":
|
||||||
|
# If source CONTRADICTS target (us), then target is contradicted
|
||||||
|
if evidence_rel == EvidenceRelationTypeV2.SUPPORTS:
|
||||||
|
evidence_rel = EvidenceRelationTypeV2.SUPPORTS
|
||||||
|
elif evidence_rel == EvidenceRelationTypeV2.CONTRADICTS:
|
||||||
|
evidence_rel = EvidenceRelationTypeV2.CONTRADICTS
|
||||||
|
# Neutral stays neutral
|
||||||
|
|
||||||
|
weight = float(rel.get("confidence", 1.0))
|
||||||
|
|
||||||
|
# Only add if there's a meaningful relationship
|
||||||
|
if rel_type_str.upper() not in ("NEUTRAL", "UNCERTAIN"):
|
||||||
|
relations.append({
|
||||||
|
"related_claim_id": target_id,
|
||||||
|
"relation_type": evidence_rel,
|
||||||
|
"weight": weight,
|
||||||
|
"reason": rel.get("reason", ""),
|
||||||
|
})
|
||||||
|
|
||||||
|
return relations
|
||||||
|
|
||||||
|
def run(self) -> dict[str, Any]:
|
||||||
|
"""Führt die vollständige Stage-8-Pipeline aus.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
dict mit 'scores', 'summary'.
|
||||||
|
"""
|
||||||
|
scores: list[dict[str, Any]] = []
|
||||||
|
errors: list[str] = []
|
||||||
|
|
||||||
|
for claim in self.claims:
|
||||||
|
claim_id = str(claim.id)
|
||||||
|
claim_text = claim.claim_text or ""
|
||||||
|
source_id = str(claim.source_id)
|
||||||
|
|
||||||
|
try:
|
||||||
|
source_data = self._get_source_data(source_id)
|
||||||
|
|
||||||
|
# --- Dimension 1: source_independence_score ---
|
||||||
|
source_independence = _compute_source_independence_score(source_data)
|
||||||
|
|
||||||
|
# --- Dimension 2: primary_source_proximity ---
|
||||||
|
proximity_score, proximity_label = _compute_primary_source_proximity(source_data)
|
||||||
|
|
||||||
|
# --- Cluster relations for this claim ---
|
||||||
|
cluster_claims, cluster_relations = self._find_claim_in_clusters(claim_id)
|
||||||
|
|
||||||
|
# --- Dimension 3: cross_source_support ---
|
||||||
|
cross_support, cross_support_data = _compute_cross_source_support(
|
||||||
|
claim_id, cluster_claims, cluster_relations, self.source_data_map
|
||||||
|
)
|
||||||
|
|
||||||
|
# --- Dimension 4: contradiction_level ---
|
||||||
|
contradiction, contradiction_data = _compute_contradiction_level(
|
||||||
|
claim_id, cluster_relations
|
||||||
|
)
|
||||||
|
|
||||||
|
# --- Dimension 5: evidence_directness ---
|
||||||
|
directness = _compute_evidence_directness(claim_text)
|
||||||
|
|
||||||
|
# --- Dimension 6: date_relevance_score ---
|
||||||
|
pub_date = getattr(claim, "event_date", None)
|
||||||
|
if pub_date is None and source_data and source_data.get("publication_date"):
|
||||||
|
pub_date = source_data.get("publication_date")
|
||||||
|
date_relevance = _compute_date_relevance_score(pub_date)
|
||||||
|
|
||||||
|
# --- evidence_type ---
|
||||||
|
evidence_type = _classify_evidence_type(claim_text)
|
||||||
|
|
||||||
|
# --- Raw scores ---
|
||||||
|
raw_scores = _build_raw_scores_json(
|
||||||
|
source_independence=source_independence,
|
||||||
|
proximity_label=proximity_label,
|
||||||
|
cross_support_data=cross_support_data,
|
||||||
|
contradiction_data=contradiction_data,
|
||||||
|
directness_score=directness,
|
||||||
|
date_score=date_relevance,
|
||||||
|
evidence_type=evidence_type.value,
|
||||||
|
claim_text=claim_text,
|
||||||
|
publication_date=pub_date,
|
||||||
|
source_data=source_data,
|
||||||
|
)
|
||||||
|
|
||||||
|
# --- Relation links ---
|
||||||
|
relation_links = self._compute_relation_links(
|
||||||
|
claim_id, claim_text, cluster_relations
|
||||||
|
)
|
||||||
|
|
||||||
|
scores.append({
|
||||||
|
"claim_id": claim_id,
|
||||||
|
"research_run_id": str(self.research_run_id),
|
||||||
|
"source_independence_score": source_independence,
|
||||||
|
"primary_source_proximity": proximity_score,
|
||||||
|
"cross_source_support": cross_support,
|
||||||
|
"contradiction_level": contradiction,
|
||||||
|
"evidence_directness": directness,
|
||||||
|
"date_relevance_score": date_relevance,
|
||||||
|
"evidence_type": evidence_type.value,
|
||||||
|
"raw_scores_json": raw_scores,
|
||||||
|
"relation_links": relation_links,
|
||||||
|
})
|
||||||
|
|
||||||
|
except Exception as exc:
|
||||||
|
logger.error("Score calculation failed for claim %s: %s", claim_id, exc)
|
||||||
|
errors.append(f"claim {claim_id}: {exc}")
|
||||||
|
|
||||||
|
# Compute summary stats
|
||||||
|
summary = self._build_summary(scores)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"scores": scores,
|
||||||
|
"summary": summary,
|
||||||
|
"errors": errors,
|
||||||
|
"research_run_id": str(self.research_run_id),
|
||||||
|
"total_claims": len(self.claims),
|
||||||
|
"scored_claims": len(scores),
|
||||||
|
}
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _build_summary(scores: list[dict[str, Any]]) -> dict[str, Any]:
|
||||||
|
"""Baut eine Zusammenfassung aller Scores."""
|
||||||
|
if not scores:
|
||||||
|
return {"total_scores": 0}
|
||||||
|
|
||||||
|
top_support = sorted(scores, key=lambda s: s["cross_source_support"], reverse=True)[:5]
|
||||||
|
top_contradiction = sorted(
|
||||||
|
scores, key=lambda s: s["contradiction_level"]
|
||||||
|
)[:5] # low = more contradiction
|
||||||
|
top_directness = sorted(scores, key=lambda s: s["evidence_directness"], reverse=True)[:5]
|
||||||
|
top_independence = sorted(scores, key=lambda s: s["source_independence_score"], reverse=True)[:5]
|
||||||
|
|
||||||
|
# High-uncertainty claims
|
||||||
|
uncertain = [
|
||||||
|
s for s in scores
|
||||||
|
if s["cross_source_support"] < 0.3
|
||||||
|
and s["contradiction_level"] < 0.5
|
||||||
|
]
|
||||||
|
|
||||||
|
# Evidence type distribution
|
||||||
|
type_dist: dict[str, int] = {}
|
||||||
|
for s in scores:
|
||||||
|
et = s["evidence_type"]
|
||||||
|
type_dist[et] = type_dist.get(et, 0) + 1
|
||||||
|
|
||||||
|
# Avg scores per dimension
|
||||||
|
avg_independence = sum(s["source_independence_score"] for s in scores) / len(scores)
|
||||||
|
avg_proximity = sum(s["primary_source_proximity"] for s in scores) / len(scores)
|
||||||
|
avg_support = sum(s["cross_source_support"] for s in scores) / len(scores)
|
||||||
|
avg_contradiction = sum(s["contradiction_level"] for s in scores) / len(scores)
|
||||||
|
avg_directness = sum(s["evidence_directness"] for s in scores) / len(scores)
|
||||||
|
avg_date = sum(s["date_relevance_score"] for s in scores) / len(scores)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"total_scores": len(scores),
|
||||||
|
"avg_source_independence": round(avg_independence, 4),
|
||||||
|
"avg_primary_source_proximity": round(avg_proximity, 4),
|
||||||
|
"avg_cross_source_support": round(avg_support, 4),
|
||||||
|
"avg_contradiction_level": round(avg_contradiction, 4),
|
||||||
|
"avg_evidence_directness": round(avg_directness, 4),
|
||||||
|
"avg_date_relevance": round(avg_date, 4),
|
||||||
|
"top_support_claims": [s["claim_id"] for s in top_support],
|
||||||
|
"top_contradiction_claims": [s["claim_id"] for s in top_contradiction],
|
||||||
|
"top_directness_claims": [s["claim_id"] for s in top_directness],
|
||||||
|
"top_independence_claims": [s["claim_id"] for s in top_independence],
|
||||||
|
"uncertain_claims": [s["claim_id"] for s in uncertain],
|
||||||
|
"uncertain_count": len(uncertain),
|
||||||
|
"evidence_type_distribution": type_dist,
|
||||||
|
}
|
||||||
@@ -316,8 +316,6 @@ class ClaimClusterModel(Base):
|
|||||||
|
|
||||||
|
|
||||||
# Association table: claims ↔ clusters (many-to-many)
|
# Association table: claims ↔ clusters (many-to-many)
|
||||||
_claim_cluster_mapping = Base() # noqa: F811 -- dummy for type resolution
|
|
||||||
|
|
||||||
claim_cluster_mapping = Base()
|
claim_cluster_mapping = Base()
|
||||||
claim_cluster_mapping.__tablename__ = "claim_cluster_mapping"
|
claim_cluster_mapping.__tablename__ = "claim_cluster_mapping"
|
||||||
claim_cluster_mapping.id = Column(String(36), primary_key=True, default=lambda: str(uuid4()))
|
claim_cluster_mapping.id = Column(String(36), primary_key=True, default=lambda: str(uuid4()))
|
||||||
@@ -380,3 +378,97 @@ class ClaimNLUModel(Base):
|
|||||||
__table_args__ = (
|
__table_args__ = (
|
||||||
Index("ix_claim_nlu_numeric_claim_id", "claim_id"),
|
Index("ix_claim_nlu_numeric_claim_id", "claim_id"),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Stage 8 — Evidence Scoring (multidimensional, transparent scores)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
class EvidenceType(str, enum.Enum):
|
||||||
|
"""Klassifizierung der Evidenz-Qualität pro Claim."""
|
||||||
|
|
||||||
|
DIRECT_OBSERVATION = "direct_observation"
|
||||||
|
SECONDARY_REPORT = "secondary_report"
|
||||||
|
ANALYSIS = "analysis"
|
||||||
|
OPINION = "opinion"
|
||||||
|
SPECULATION = "speculation"
|
||||||
|
|
||||||
|
|
||||||
|
class EvidenceRelationTypeV2(str, enum.Enum):
|
||||||
|
"""Relation zwischen einem Scored-Claim und einem anderen Claim (Stage 8)."""
|
||||||
|
|
||||||
|
SUPPORTS = "supports"
|
||||||
|
CONTRADICTS = "contradicts"
|
||||||
|
NEUTRAL = "neutral"
|
||||||
|
|
||||||
|
|
||||||
|
class EvidenceScoreModel(Base):
|
||||||
|
"""Transparente multidimensionale Scores für jeden Claim (Stage 8)."""
|
||||||
|
|
||||||
|
__tablename__ = "evidence_scores"
|
||||||
|
|
||||||
|
id = Column(String(36), primary_key=True, default=lambda: str(uuid4()))
|
||||||
|
claim_id = Column(String(36), ForeignKey("claims.id"), nullable=False, unique=True)
|
||||||
|
research_run_id = Column(String(36), nullable=False)
|
||||||
|
|
||||||
|
# Dimension 1: source independence
|
||||||
|
source_independence_score = Column(Float, nullable=False, default=0.5)
|
||||||
|
# Dimension 2: proximity to primary source
|
||||||
|
primary_source_proximity = Column(Float, nullable=False, default=0.0)
|
||||||
|
# Dimension 3: cross-source support
|
||||||
|
cross_source_support = Column(Float, nullable=False, default=0.0)
|
||||||
|
# Dimension 4: contradiction level (1.0 = no contradictions)
|
||||||
|
contradiction_level = Column(Float, nullable=False, default=1.0)
|
||||||
|
# Dimension 5: directness of evidence
|
||||||
|
evidence_directness = Column(Float, nullable=False, default=0.5)
|
||||||
|
# Dimension 6: date relevance
|
||||||
|
date_relevance_score = Column(Float, nullable=False, default=0.5)
|
||||||
|
|
||||||
|
# Evidence classification
|
||||||
|
evidence_type = Column(Enum(EvidenceType), nullable=False, default=EvidenceType.SECONDARY_REPORT)
|
||||||
|
|
||||||
|
# Raw scores for full auditability
|
||||||
|
raw_scores_json = Column(JSON, nullable=False, default=dict)
|
||||||
|
|
||||||
|
created_at = Column(DateTime, nullable=False, default=datetime.utcnow)
|
||||||
|
updated_at = Column(DateTime, nullable=False, default=datetime.utcnow)
|
||||||
|
|
||||||
|
# Relationships
|
||||||
|
relations = relationship(
|
||||||
|
"EvidenceScoreRelationModel",
|
||||||
|
back_populates="score",
|
||||||
|
cascade="all, delete-orphan",
|
||||||
|
foreign_keys="EvidenceScoreRelationModel.score_id",
|
||||||
|
)
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
Index("ix_evidence_scores_claim_id", "claim_id"),
|
||||||
|
Index("ix_evidence_scores_research_run_id", "research_run_id"),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class EvidenceScoreRelationModel(Base):
|
||||||
|
"""Relation zwischen einem Evidence-Scored Claim und anderen Claims (Stage 8)."""
|
||||||
|
|
||||||
|
__tablename__ = "evidence_score_relations"
|
||||||
|
|
||||||
|
id = Column(String(36), primary_key=True, default=lambda: str(uuid4()))
|
||||||
|
score_id = Column(
|
||||||
|
String(36),
|
||||||
|
ForeignKey("evidence_scores.id"),
|
||||||
|
nullable=False,
|
||||||
|
)
|
||||||
|
related_claim_id = Column(String(36), ForeignKey("claims.id"), nullable=False)
|
||||||
|
relation_type = Column(Enum(EvidenceRelationTypeV2), nullable=False)
|
||||||
|
weight = Column(Float, nullable=False, default=1.0)
|
||||||
|
created_at = Column(DateTime, nullable=False, default=datetime.utcnow)
|
||||||
|
|
||||||
|
# Relationships
|
||||||
|
score = relationship("EvidenceScoreModel", back_populates="relations", foreign_keys=[score_id])
|
||||||
|
related_claim = relationship("ClaimModel", foreign_keys=[related_claim_id])
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
Index("ix_evidence_score_relations_score_id", "score_id"),
|
||||||
|
Index("ix_evidence_score_relations_related_claim_id", "related_claim_id"),
|
||||||
|
)
|
||||||
1022
tests/stages/test_stage8_evidence_scoring.py
Normal file
1022
tests/stages/test_stage8_evidence_scoring.py
Normal file
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user