thetruthbureau / verifier.py
nahArnav's picture
Update verifier.py
1b7f23c verified
Raw History Blame Contribute Delete
12.7 kB
"""
Internet Verifier for VeriLens AI
- Searches the web via Google News RSS for live, rate-limit-proof verification.
- Searches Wikipedia API for historical fact verification.
- Computes strict semantic entailment using an NLI Cross-Encoder with Fact-Check Guardrails.
"""
from __future__ import annotations
import urllib.request
import urllib.parse
import xml.etree.ElementTree as ET
import re
import json
import html # <-- For removing ugly tags
import numpy as np
import asyncio
import logging
from dataclasses import dataclass, field
logger = logging.getLogger(__name__)
# ── Lazy-loaded Cross-Encoder ────────────────────────────────────────
_cross_model = None
def _get_cross_model():
global _cross_model
if _cross_model is None:
try:
from sentence_transformers import CrossEncoder
logger.info("Loading Multilingual NLI Cross-Encoder model…")
_cross_model = CrossEncoder("MoritzLaurer/mDeBERTa-v3-base-xnli-multilingual-nli-2mil7")
logger.info("Multilingual NLI Cross-Encoder loaded successfully.")
except Exception as exc:
logger.warning("Could not load NLI Cross-Encoder: %s", exc)
return _cross_model
@dataclass
class SourceArticle:
title: str
url: str
snippet: str
trust: str = "medium"
@dataclass
class VerificationResult:
similarity_score: float = 0.0
sources: list[SourceArticle] = field(default_factory=list)
verified: bool = False
max_entailment: float = 0.0
# ── Trusted domains (Expanded Global & Indian Scope) ─────────────────
HIGH_TRUST_DOMAINS = {
"wikipedia.org", "reuters.com", "apnews.com", "bloomberg.com", "afp.com", "upi.com",
"bbc.com", "bbc.co.uk", "nytimes.com", "washingtonpost.com", "wsj.com",
"theguardian.com", "npr.org", "pbs.org", "cnn.com", "ft.com",
"aljazeera.com", "dw.com", "france24.com", "scmp.com", "nbcnews.com",
"cbsnews.com", "abcnews.go.com", "theatlantic.com", "time.com", "economist.com",
"thehindu.com", "hindustantimes.com", "indianexpress.com", "timesofindia.indiatimes.com",
"ndtv.com", "indiatoday.in", "theprint.in", "thewire.in", "scroll.in",
"livemint.com", "business-standard.com", "deccanherald.com", "telegraphindia.com",
"tribuneindia.com", "newindianexpress.com", "firstpost.com", "thequint.com",
"cnbctv18.com", "moneycontrol.com", "aninews.in", "ptinews.com", "freepressjournal.in",
"snopes.com", "politifact.com", "factcheck.org", "altnews.in", "boomlive.in",
"newschecker.in", "vishvasnews.com", "smhoaxinvestigator.com", "factchecker.in",
"yahoo.com/news", "msn.com", "news.google.com"
}
LOW_TRUST_DOMAINS = {
"infowars.com", "naturalnews.com", "beforeitsnews.com", "thegatewaypundit.com",
"zerohedge.com", "worldnewsdailyreport.com", "nationalreport.net",
"rt.com", "sputniknews.com", "globaltimes.cn",
"postcard.news", "opindia.com", "tfipost.com", "kreately.in", "rightlog.in",
"theonion.com", "babylonbee.com", "fakingnews.com", "thefauxy.com",
"thedailymash.co.uk", "waterfordwhispersnews.com", "clickhole.com"
}
def _trust_level(url: str, snippet: str = "", title: str = "") -> str:
lower_url = url.lower()
lower_snippet = snippet.lower()
lower_title = title.lower()
for d in HIGH_TRUST_DOMAINS:
if d in lower_url: return "high"
high_trust_keywords = ["reuters", "associated press", "bbc", "cnn", "the new york times", "bloomberg"]
for keyword in high_trust_keywords:
if keyword in lower_snippet or keyword in lower_title: return "high"
for d in LOW_TRUST_DOMAINS:
if d in lower_url: return "low"
return "medium"
# ── Locale detection for multilingual search ─────────────────────────
_LOCALE_MAP = {
(0x0900, 0x097F): ('hi', 'IN'), (0x0980, 0x09FF): ('bn', 'IN'),
(0x0A00, 0x0A7F): ('pa', 'IN'), (0x0A80, 0x0AFF): ('gu', 'IN'),
(0x0B80, 0x0BFF): ('ta', 'IN'), (0x0C00, 0x0C7F): ('te', 'IN'),
(0x0C80, 0x0CFF): ('kn', 'IN'), (0x0D00, 0x0D7F): ('ml', 'IN'),
(0x0600, 0x06FF): ('ar', 'AE'), (0x4E00, 0x9FFF): ('zh', 'CN'),
(0x3040, 0x30FF): ('ja', 'JP'), (0xAC00, 0xD7AF): ('ko', 'KR'),
(0x0400, 0x04FF): ('ru', 'RU'),
}
def _detect_locale(query: str) -> tuple[str, str]:
for c in query:
cp = ord(c)
for (lo, hi), locale in _LOCALE_MAP.items():
if lo <= cp <= hi: return locale
return ('en', 'US')
def _fetch_google_rss(url: str, num_results: int) -> list[dict]:
req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0'})
try:
with urllib.request.urlopen(req, timeout=10) as response:
xml_data = response.read()
root = ET.fromstring(xml_data)
results = []
for item in root.findall('.//item')[:num_results]:
title = item.find('title')
link = item.find('link')
desc = item.find('description')
desc_html = desc.text if desc is not None else ""
snippet = re.sub('<[^<]+>', '', desc_html)
results.append({
"title": html.unescape(title.text) if title is not None else "",
"href": link.text if link is not None else "",
"body": html.unescape(snippet)
})
return results
except Exception as e:
logger.error(f"Google RSS Error: {e}")
return []
def _google_news_search(query: str, num_results: int = 8) -> list[dict]:
try:
safe_query = urllib.parse.quote(query)
lang, country = _detect_locale(query)
url = f"https://news.google.com/rss/search?q={safe_query}&hl={lang}&gl={country}&ceid={country}:{lang}"
results = _fetch_google_rss(url, num_results)
if not results:
url_nolang = f"https://news.google.com/rss/search?q={safe_query}"
results = _fetch_google_rss(url_nolang, num_results)
if not results:
words = query.split()
if len(words) > 4:
short_query = " ".join(words[:6])
url_short = f"https://news.google.com/rss/search?q={urllib.parse.quote(short_query)}&hl={lang}&gl={country}&ceid={country}:{lang}"
results = _fetch_google_rss(url_short, num_results)
return results
except Exception as exc:
logger.error("Google News search failed: %s", exc)
return []
def _wikipedia_search(query: str) -> list[dict]:
def _wiki_query(wiki_lang: str, q: str) -> list[dict]:
safe_query = urllib.parse.quote(q)
url = f"https://{wiki_lang}.wikipedia.org/w/api.php?action=query&generator=search&gsrsearch={safe_query}&gsrlimit=2&prop=extracts&exchars=350&explaintext=1&utf8=&format=json"
req = urllib.request.Request(url, headers={'User-Agent': 'VeriLensAI/1.0'})
try:
with urllib.request.urlopen(req, timeout=10) as response:
data = json.loads(response.read().decode())
results = []
pages = data.get('query', {}).get('pages', {})
for page_id, item in pages.items():
title = item.get('title', '')
clean_snippet = item.get('extract', '').replace('\n', ' ')
if clean_snippet:
results.append({
"title": html.unescape(f"{title} - Wikipedia"),
"href": f"https://{wiki_lang}.wikipedia.org/wiki/{urllib.parse.quote(title.replace(' ', '_'))}",
"body": html.unescape(clean_snippet)
})
return results
except Exception as e:
logger.error(f"Wiki Query Error: {e}")
return []
try:
results = _wiki_query('en', query)
if not results and any(ord(c) > 127 for c in query):
detected_lang, _ = _detect_locale(query)
if detected_lang != 'en': results = _wiki_query(detected_lang, query)
return results
except Exception as exc:
logger.error("Wikipedia search failed: %s", exc)
return []
async def _search_web(query: str, num_results: int = 8) -> list[dict]:
news_task = asyncio.to_thread(_google_news_search, query, num_results)
wiki_task = asyncio.to_thread(_wikipedia_search, query)
news_results, wiki_results = await asyncio.gather(news_task, wiki_task)
half_quota = num_results // 2
balanced_results = news_results[:half_quota] + wiki_results[:num_results - half_quota]
if len(balanced_results) < num_results:
remaining_slots = num_results - len(balanced_results)
balanced_results.extend(news_results[half_quota:half_quota + remaining_slots])
return balanced_results
# ── NLI Entailment & Guardrails ──────────────────────────────────────
MIN_RELEVANCE_THRESHOLD = 0.75
_NLI_ENTAILMENT_IDX = 0
def _softmax(logits: np.ndarray) -> np.ndarray:
exp = np.exp(logits - np.max(logits, axis=-1, keepdims=True))
return exp / np.sum(exp, axis=-1, keepdims=True)
def _compute_per_source_similarity(text: str, snippets: list[str]) -> list[float]:
model = _get_cross_model()
if model is None or not snippets:
return [0.0] * len(snippets)
try:
pairs = [[snippet[:512], text[:512]] for snippet in snippets]
logits = model.predict(pairs)
logits = np.array(logits)
if logits.ndim == 1: logits = logits.reshape(1, -1)
probs = _softmax(logits)
entailment_scores = probs[:, _NLI_ENTAILMENT_IDX]
# πŸ›‘οΈ THE FACT-CHECK GUARDRAIL πŸ›‘οΈ
# Intercepts Lexical Overlap Bias. If the article is actively debunking the claim,
# we force the NLI entailment score to 0.0 so it is rejected.
debunk_keywords = ["fake", "conspiracy", "debunk", "hoax", "false", "untrue", "fact check", "fact-check", "misinformation", "rumor"]
final_scores = []
for i, score in enumerate(entailment_scores):
snippet_lower = snippets[i].lower()
if any(word in snippet_lower for word in debunk_keywords):
print("πŸ›‘ GUARDRAIL TRIGGERED: Debunking keyword found in source. Forcing score to 0.0")
final_scores.append(0.0)
else:
final_scores.append(float(score))
return final_scores
except Exception as exc:
logger.error("NLI entailment computation failed: %s", exc)
return [0.0] * len(snippets)
async def verify_claim(text: str, search_query: str) -> VerificationResult:
items = await _search_web(search_query)
if not items:
return VerificationResult(similarity_score=0.0, sources=[], verified=False)
candidates: list[SourceArticle] = []
snippets: list[str] = []
for item in items:
title = item.get("title", "")
link = item.get("url", "") or item.get("href", "")
snippet = item.get("body", "")
candidates.append(
SourceArticle(
title=title, url=link, snippet=snippet,
trust=_trust_level(url=link, snippet=snippet, title=title),
)
)
snippets.append(f"{title}. {snippet}")
scores = await asyncio.to_thread(_compute_per_source_similarity, text, snippets)
sources: list[SourceArticle] = []
relevant_scores: list[float] = []
print("\n" + "="*50)
print("🧠 CROSS-ENCODER SCORES:")
for candidate, score in zip(candidates, scores):
required_score = 0.45 if "wikipedia.org" in candidate.url else MIN_RELEVANCE_THRESHOLD
if score >= required_score:
sources.append(candidate)
relevant_scores.append(score)
print(f" -> βœ… ACCEPTED ({score:.3f}) | {candidate.url}")
else:
print(f" -> ❌ REJECTED ({score:.3f}) | {candidate.url}")
print("="*50 + "\n")
# πŸ›‘οΈ THE BUG FIX: If no sources passed the threshold, it safely returns FALSE.
if not sources:
return VerificationResult(similarity_score=0.0, sources=[], verified=False)
avg_similarity = sum(relevant_scores) / len(relevant_scores)
peak_entailment = max(relevant_scores)
return VerificationResult(
similarity_score=round(avg_similarity, 4),
sources=sources,
verified=True, # Only returns True if valid, non-debunking sources survived
max_entailment=round(peak_entailment, 4),
)