Spaces:
Sleeping
Sleeping
Download verifier.py from nahArnav/thetruthbureau: direct link, hf CLI and curl.
- Browser
- Download file 12.7 kB
-
https://huggingface.co/spaces/nahArnav/thetruthbureau/resolve/main/verifier.py
- Command line
-
hf download hf://spaces/nahArnav/thetruthbureau/verifier.py
-
curl -L -o verifier.py https://huggingface.co/spaces/nahArnav/thetruthbureau/resolve/main/verifier.py
12.7 kB
| """ | |
| Internet Verifier for VeriLens AI | |
| - Searches the web via Google News RSS for live, rate-limit-proof verification. | |
| - Searches Wikipedia API for historical fact verification. | |
| - Computes strict semantic entailment using an NLI Cross-Encoder with Fact-Check Guardrails. | |
| """ | |
| from __future__ import annotations | |
| import urllib.request | |
| import urllib.parse | |
| import xml.etree.ElementTree as ET | |
| import re | |
| import json | |
| import html # <-- For removing ugly tags | |
| import numpy as np | |
| import asyncio | |
| import logging | |
| from dataclasses import dataclass, field | |
| logger = logging.getLogger(__name__) | |
| # ββ Lazy-loaded Cross-Encoder ββββββββββββββββββββββββββββββββββββββββ | |
| _cross_model = None | |
| def _get_cross_model(): | |
| global _cross_model | |
| if _cross_model is None: | |
| try: | |
| from sentence_transformers import CrossEncoder | |
| logger.info("Loading Multilingual NLI Cross-Encoder modelβ¦") | |
| _cross_model = CrossEncoder("MoritzLaurer/mDeBERTa-v3-base-xnli-multilingual-nli-2mil7") | |
| logger.info("Multilingual NLI Cross-Encoder loaded successfully.") | |
| except Exception as exc: | |
| logger.warning("Could not load NLI Cross-Encoder: %s", exc) | |
| return _cross_model | |
| class SourceArticle: | |
| title: str | |
| url: str | |
| snippet: str | |
| trust: str = "medium" | |
| class VerificationResult: | |
| similarity_score: float = 0.0 | |
| sources: list[SourceArticle] = field(default_factory=list) | |
| verified: bool = False | |
| max_entailment: float = 0.0 | |
| # ββ Trusted domains (Expanded Global & Indian Scope) βββββββββββββββββ | |
| HIGH_TRUST_DOMAINS = { | |
| "wikipedia.org", "reuters.com", "apnews.com", "bloomberg.com", "afp.com", "upi.com", | |
| "bbc.com", "bbc.co.uk", "nytimes.com", "washingtonpost.com", "wsj.com", | |
| "theguardian.com", "npr.org", "pbs.org", "cnn.com", "ft.com", | |
| "aljazeera.com", "dw.com", "france24.com", "scmp.com", "nbcnews.com", | |
| "cbsnews.com", "abcnews.go.com", "theatlantic.com", "time.com", "economist.com", | |
| "thehindu.com", "hindustantimes.com", "indianexpress.com", "timesofindia.indiatimes.com", | |
| "ndtv.com", "indiatoday.in", "theprint.in", "thewire.in", "scroll.in", | |
| "livemint.com", "business-standard.com", "deccanherald.com", "telegraphindia.com", | |
| "tribuneindia.com", "newindianexpress.com", "firstpost.com", "thequint.com", | |
| "cnbctv18.com", "moneycontrol.com", "aninews.in", "ptinews.com", "freepressjournal.in", | |
| "snopes.com", "politifact.com", "factcheck.org", "altnews.in", "boomlive.in", | |
| "newschecker.in", "vishvasnews.com", "smhoaxinvestigator.com", "factchecker.in", | |
| "yahoo.com/news", "msn.com", "news.google.com" | |
| } | |
| LOW_TRUST_DOMAINS = { | |
| "infowars.com", "naturalnews.com", "beforeitsnews.com", "thegatewaypundit.com", | |
| "zerohedge.com", "worldnewsdailyreport.com", "nationalreport.net", | |
| "rt.com", "sputniknews.com", "globaltimes.cn", | |
| "postcard.news", "opindia.com", "tfipost.com", "kreately.in", "rightlog.in", | |
| "theonion.com", "babylonbee.com", "fakingnews.com", "thefauxy.com", | |
| "thedailymash.co.uk", "waterfordwhispersnews.com", "clickhole.com" | |
| } | |
| def _trust_level(url: str, snippet: str = "", title: str = "") -> str: | |
| lower_url = url.lower() | |
| lower_snippet = snippet.lower() | |
| lower_title = title.lower() | |
| for d in HIGH_TRUST_DOMAINS: | |
| if d in lower_url: return "high" | |
| high_trust_keywords = ["reuters", "associated press", "bbc", "cnn", "the new york times", "bloomberg"] | |
| for keyword in high_trust_keywords: | |
| if keyword in lower_snippet or keyword in lower_title: return "high" | |
| for d in LOW_TRUST_DOMAINS: | |
| if d in lower_url: return "low" | |
| return "medium" | |
| # ββ Locale detection for multilingual search βββββββββββββββββββββββββ | |
| _LOCALE_MAP = { | |
| (0x0900, 0x097F): ('hi', 'IN'), (0x0980, 0x09FF): ('bn', 'IN'), | |
| (0x0A00, 0x0A7F): ('pa', 'IN'), (0x0A80, 0x0AFF): ('gu', 'IN'), | |
| (0x0B80, 0x0BFF): ('ta', 'IN'), (0x0C00, 0x0C7F): ('te', 'IN'), | |
| (0x0C80, 0x0CFF): ('kn', 'IN'), (0x0D00, 0x0D7F): ('ml', 'IN'), | |
| (0x0600, 0x06FF): ('ar', 'AE'), (0x4E00, 0x9FFF): ('zh', 'CN'), | |
| (0x3040, 0x30FF): ('ja', 'JP'), (0xAC00, 0xD7AF): ('ko', 'KR'), | |
| (0x0400, 0x04FF): ('ru', 'RU'), | |
| } | |
| def _detect_locale(query: str) -> tuple[str, str]: | |
| for c in query: | |
| cp = ord(c) | |
| for (lo, hi), locale in _LOCALE_MAP.items(): | |
| if lo <= cp <= hi: return locale | |
| return ('en', 'US') | |
| def _fetch_google_rss(url: str, num_results: int) -> list[dict]: | |
| req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0'}) | |
| try: | |
| with urllib.request.urlopen(req, timeout=10) as response: | |
| xml_data = response.read() | |
| root = ET.fromstring(xml_data) | |
| results = [] | |
| for item in root.findall('.//item')[:num_results]: | |
| title = item.find('title') | |
| link = item.find('link') | |
| desc = item.find('description') | |
| desc_html = desc.text if desc is not None else "" | |
| snippet = re.sub('<[^<]+>', '', desc_html) | |
| results.append({ | |
| "title": html.unescape(title.text) if title is not None else "", | |
| "href": link.text if link is not None else "", | |
| "body": html.unescape(snippet) | |
| }) | |
| return results | |
| except Exception as e: | |
| logger.error(f"Google RSS Error: {e}") | |
| return [] | |
| def _google_news_search(query: str, num_results: int = 8) -> list[dict]: | |
| try: | |
| safe_query = urllib.parse.quote(query) | |
| lang, country = _detect_locale(query) | |
| url = f"https://news.google.com/rss/search?q={safe_query}&hl={lang}&gl={country}&ceid={country}:{lang}" | |
| results = _fetch_google_rss(url, num_results) | |
| if not results: | |
| url_nolang = f"https://news.google.com/rss/search?q={safe_query}" | |
| results = _fetch_google_rss(url_nolang, num_results) | |
| if not results: | |
| words = query.split() | |
| if len(words) > 4: | |
| short_query = " ".join(words[:6]) | |
| url_short = f"https://news.google.com/rss/search?q={urllib.parse.quote(short_query)}&hl={lang}&gl={country}&ceid={country}:{lang}" | |
| results = _fetch_google_rss(url_short, num_results) | |
| return results | |
| except Exception as exc: | |
| logger.error("Google News search failed: %s", exc) | |
| return [] | |
| def _wikipedia_search(query: str) -> list[dict]: | |
| def _wiki_query(wiki_lang: str, q: str) -> list[dict]: | |
| safe_query = urllib.parse.quote(q) | |
| url = f"https://{wiki_lang}.wikipedia.org/w/api.php?action=query&generator=search&gsrsearch={safe_query}&gsrlimit=2&prop=extracts&exchars=350&explaintext=1&utf8=&format=json" | |
| req = urllib.request.Request(url, headers={'User-Agent': 'VeriLensAI/1.0'}) | |
| try: | |
| with urllib.request.urlopen(req, timeout=10) as response: | |
| data = json.loads(response.read().decode()) | |
| results = [] | |
| pages = data.get('query', {}).get('pages', {}) | |
| for page_id, item in pages.items(): | |
| title = item.get('title', '') | |
| clean_snippet = item.get('extract', '').replace('\n', ' ') | |
| if clean_snippet: | |
| results.append({ | |
| "title": html.unescape(f"{title} - Wikipedia"), | |
| "href": f"https://{wiki_lang}.wikipedia.org/wiki/{urllib.parse.quote(title.replace(' ', '_'))}", | |
| "body": html.unescape(clean_snippet) | |
| }) | |
| return results | |
| except Exception as e: | |
| logger.error(f"Wiki Query Error: {e}") | |
| return [] | |
| try: | |
| results = _wiki_query('en', query) | |
| if not results and any(ord(c) > 127 for c in query): | |
| detected_lang, _ = _detect_locale(query) | |
| if detected_lang != 'en': results = _wiki_query(detected_lang, query) | |
| return results | |
| except Exception as exc: | |
| logger.error("Wikipedia search failed: %s", exc) | |
| return [] | |
| async def _search_web(query: str, num_results: int = 8) -> list[dict]: | |
| news_task = asyncio.to_thread(_google_news_search, query, num_results) | |
| wiki_task = asyncio.to_thread(_wikipedia_search, query) | |
| news_results, wiki_results = await asyncio.gather(news_task, wiki_task) | |
| half_quota = num_results // 2 | |
| balanced_results = news_results[:half_quota] + wiki_results[:num_results - half_quota] | |
| if len(balanced_results) < num_results: | |
| remaining_slots = num_results - len(balanced_results) | |
| balanced_results.extend(news_results[half_quota:half_quota + remaining_slots]) | |
| return balanced_results | |
| # ββ NLI Entailment & Guardrails ββββββββββββββββββββββββββββββββββββββ | |
| MIN_RELEVANCE_THRESHOLD = 0.75 | |
| _NLI_ENTAILMENT_IDX = 0 | |
| def _softmax(logits: np.ndarray) -> np.ndarray: | |
| exp = np.exp(logits - np.max(logits, axis=-1, keepdims=True)) | |
| return exp / np.sum(exp, axis=-1, keepdims=True) | |
| def _compute_per_source_similarity(text: str, snippets: list[str]) -> list[float]: | |
| model = _get_cross_model() | |
| if model is None or not snippets: | |
| return [0.0] * len(snippets) | |
| try: | |
| pairs = [[snippet[:512], text[:512]] for snippet in snippets] | |
| logits = model.predict(pairs) | |
| logits = np.array(logits) | |
| if logits.ndim == 1: logits = logits.reshape(1, -1) | |
| probs = _softmax(logits) | |
| entailment_scores = probs[:, _NLI_ENTAILMENT_IDX] | |
| # π‘οΈ THE FACT-CHECK GUARDRAIL π‘οΈ | |
| # Intercepts Lexical Overlap Bias. If the article is actively debunking the claim, | |
| # we force the NLI entailment score to 0.0 so it is rejected. | |
| debunk_keywords = ["fake", "conspiracy", "debunk", "hoax", "false", "untrue", "fact check", "fact-check", "misinformation", "rumor"] | |
| final_scores = [] | |
| for i, score in enumerate(entailment_scores): | |
| snippet_lower = snippets[i].lower() | |
| if any(word in snippet_lower for word in debunk_keywords): | |
| print("π GUARDRAIL TRIGGERED: Debunking keyword found in source. Forcing score to 0.0") | |
| final_scores.append(0.0) | |
| else: | |
| final_scores.append(float(score)) | |
| return final_scores | |
| except Exception as exc: | |
| logger.error("NLI entailment computation failed: %s", exc) | |
| return [0.0] * len(snippets) | |
| async def verify_claim(text: str, search_query: str) -> VerificationResult: | |
| items = await _search_web(search_query) | |
| if not items: | |
| return VerificationResult(similarity_score=0.0, sources=[], verified=False) | |
| candidates: list[SourceArticle] = [] | |
| snippets: list[str] = [] | |
| for item in items: | |
| title = item.get("title", "") | |
| link = item.get("url", "") or item.get("href", "") | |
| snippet = item.get("body", "") | |
| candidates.append( | |
| SourceArticle( | |
| title=title, url=link, snippet=snippet, | |
| trust=_trust_level(url=link, snippet=snippet, title=title), | |
| ) | |
| ) | |
| snippets.append(f"{title}. {snippet}") | |
| scores = await asyncio.to_thread(_compute_per_source_similarity, text, snippets) | |
| sources: list[SourceArticle] = [] | |
| relevant_scores: list[float] = [] | |
| print("\n" + "="*50) | |
| print("π§ CROSS-ENCODER SCORES:") | |
| for candidate, score in zip(candidates, scores): | |
| required_score = 0.45 if "wikipedia.org" in candidate.url else MIN_RELEVANCE_THRESHOLD | |
| if score >= required_score: | |
| sources.append(candidate) | |
| relevant_scores.append(score) | |
| print(f" -> β ACCEPTED ({score:.3f}) | {candidate.url}") | |
| else: | |
| print(f" -> β REJECTED ({score:.3f}) | {candidate.url}") | |
| print("="*50 + "\n") | |
| # π‘οΈ THE BUG FIX: If no sources passed the threshold, it safely returns FALSE. | |
| if not sources: | |
| return VerificationResult(similarity_score=0.0, sources=[], verified=False) | |
| avg_similarity = sum(relevant_scores) / len(relevant_scores) | |
| peak_entailment = max(relevant_scores) | |
| return VerificationResult( | |
| similarity_score=round(avg_similarity, 4), | |
| sources=sources, | |
| verified=True, # Only returns True if valid, non-debunking sources survived | |
| max_entailment=round(peak_entailment, 4), | |
| ) |