## Copyright © 2026 Olaf Kolkman
## SPDX-License-Identifier: GPL-3.0-or-later
import httpx
import logging
from html.parser import HTMLParser
from urllib.parse import urlparse
logger = logging.getLogger(__name__)
class TitleParser(HTMLParser):
def __init__(self):
super().__init__()
self.title = None
self.og_title = None
self.twitter_title = None
self.meta_title = None
self.in_title = False
def handle_starttag(self, tag, attrs):
if tag.lower() == 'title':
self.in_title = True
elif tag.lower() == 'meta':
attrs_dict = dict(attrs)
# Check for Open Graph title
if attrs_dict.get('property', '').lower() == 'og:title':
content = attrs_dict.get('content', '').strip()
if content and not self.og_title:
self.og_title = content
# Check for Twitter title
elif attrs_dict.get('name', '').lower() == 'twitter:title':
content = attrs_dict.get('content', '').strip()
if content and not self.twitter_title:
self.twitter_title = content
# Check for generic meta title
elif attrs_dict.get('name', '').lower() == 'title':
content = attrs_dict.get('content', '').strip()
if content and not self.meta_title:
self.meta_title = content
def handle_endtag(self, tag):
if tag.lower() == 'title':
self.in_title = False
def handle_data(self, data):
if self.in_title and not self.title:
stripped = data.strip()
if stripped:
self.title = stripped
def get_best_title(self):
"""Return the best title found, matching browser behavior.
Priority: page
tag (what browser shows), then meta tags as fallback."""
return self.title or self.og_title or self.twitter_title or self.meta_title
def scrape_title(url: str) -> str:
"""
Scrape the title from a URL, checking multiple sources:
1. Page title tag (what browser shows)
2. Open Graph title (og:title meta tag)
3. Twitter title (twitter:title meta tag)
4. Generic meta title
5. Domain name as fallback
"""
try:
# Parse URL to extract domain as fallback
parsed = urlparse(url)
domain = parsed.netloc or url
# Fetch the URL with a timeout and size limit, using iter_bytes for decompression
with httpx.stream('GET', url, follow_redirects=True, timeout=5.0) as response:
if response.status_code != 200:
logger.warning('Failed to fetch %s: status %d', url, response.status_code)
return domain
# Read HTML in chunks (auto-decompressed) to avoid loading huge files
html_content = b''
max_size = 1024 * 100 # 100 KB limit
for chunk in response.iter_bytes():
html_content += chunk
if len(html_content) > max_size:
break
# Parse the HTML to extract title
try:
html_text = html_content.decode('utf-8', errors='ignore')
parser = TitleParser()
parser.feed(html_text)
best_title = parser.get_best_title()
if best_title:
return best_title
except Exception as e:
logger.warning('Failed to parse HTML from %s: %s', url, e)
return domain
except httpx.TimeoutException:
logger.warning('Timeout fetching %s', url)
return urlparse(url).netloc or url
except httpx.NetworkError as e:
logger.warning('Network error fetching %s: %s', url, e)
return urlparse(url).netloc or url
except Exception as e:
logger.error('Unexpected error scraping %s: %s', url, e)
return urlparse(url).netloc or url