Add media URL extractors for Reddit, Imgur, and Gfycat
This commit is contained in:
parent
4e6a186484
commit
1484c1f135
4 changed files with 159 additions and 0 deletions
37
src/extractors/__init__.py
Normal file
37
src/extractors/__init__.py
Normal file
|
|
@ -0,0 +1,37 @@
|
||||||
|
"""Media URL extractors for various hosts."""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from typing import Optional
|
||||||
|
from urllib.parse import urlparse
|
||||||
|
|
||||||
|
from .imgur import extract_imgur_url
|
||||||
|
from .reddit import extract_reddit_video_url
|
||||||
|
from .gfycat import extract_gfycat_url
|
||||||
|
|
||||||
|
logger = logging.getLogger("reddit_collector")
|
||||||
|
|
||||||
|
|
||||||
|
def extract_media_url(url: str, media_type: str) -> tuple[str, str]:
|
||||||
|
"""
|
||||||
|
Extract the actual downloadable media URL from a post URL.
|
||||||
|
Returns (final_url, final_media_type).
|
||||||
|
"""
|
||||||
|
parsed = urlparse(url)
|
||||||
|
domain = parsed.netloc.lower()
|
||||||
|
|
||||||
|
if "v.redd.it" in domain:
|
||||||
|
video_url = extract_reddit_video_url(url)
|
||||||
|
if video_url:
|
||||||
|
return video_url, "video"
|
||||||
|
|
||||||
|
if "imgur.com" in domain:
|
||||||
|
imgur_url, imgur_type = extract_imgur_url(url)
|
||||||
|
if imgur_url:
|
||||||
|
return imgur_url, imgur_type
|
||||||
|
|
||||||
|
if "gfycat.com" in domain or "redgifs.com" in domain:
|
||||||
|
gfycat_url = extract_gfycat_url(url)
|
||||||
|
if gfycat_url:
|
||||||
|
return gfycat_url, "video"
|
||||||
|
|
||||||
|
return url, media_type
|
||||||
44
src/extractors/gfycat.py
Normal file
44
src/extractors/gfycat.py
Normal file
|
|
@ -0,0 +1,44 @@
|
||||||
|
"""Gfycat/Redgifs URL extractor using yt-dlp."""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from typing import Optional
|
||||||
|
|
||||||
|
logger = logging.getLogger("reddit_collector")
|
||||||
|
|
||||||
|
|
||||||
|
def extract_gfycat_url(url: str) -> Optional[str]:
|
||||||
|
"""
|
||||||
|
Extract video URL from Gfycat/Redgifs links.
|
||||||
|
Uses yt-dlp for extraction.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
import yt_dlp
|
||||||
|
|
||||||
|
ydl_opts = {
|
||||||
|
"quiet": True,
|
||||||
|
"no_warnings": True,
|
||||||
|
}
|
||||||
|
|
||||||
|
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||||||
|
info = ydl.extract_info(url, download=False)
|
||||||
|
|
||||||
|
if info and "url" in info:
|
||||||
|
return info["url"]
|
||||||
|
|
||||||
|
if info and "formats" in info:
|
||||||
|
mp4_formats = [
|
||||||
|
f for f in info["formats"]
|
||||||
|
if f.get("ext") == "mp4" and f.get("url")
|
||||||
|
]
|
||||||
|
if mp4_formats:
|
||||||
|
best = max(mp4_formats, key=lambda x: x.get("height", 0))
|
||||||
|
return best["url"]
|
||||||
|
|
||||||
|
return None
|
||||||
|
|
||||||
|
except ImportError:
|
||||||
|
logger.warning("yt-dlp not installed, cannot extract Gfycat URLs")
|
||||||
|
return None
|
||||||
|
except Exception as e:
|
||||||
|
logger.debug(f"Failed to extract Gfycat URL: {e}")
|
||||||
|
return None
|
||||||
33
src/extractors/imgur.py
Normal file
33
src/extractors/imgur.py
Normal file
|
|
@ -0,0 +1,33 @@
|
||||||
|
"""Imgur URL extractor."""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import re
|
||||||
|
from typing import Optional
|
||||||
|
from urllib.parse import urlparse
|
||||||
|
|
||||||
|
logger = logging.getLogger("reddit_collector")
|
||||||
|
|
||||||
|
|
||||||
|
def extract_imgur_url(url: str) -> tuple[Optional[str], str]:
|
||||||
|
"""
|
||||||
|
Extract direct image/video URL from Imgur links.
|
||||||
|
Returns (url, media_type) or (None, "image") on failure.
|
||||||
|
"""
|
||||||
|
parsed = urlparse(url)
|
||||||
|
path = parsed.path
|
||||||
|
|
||||||
|
if "i.imgur.com" in parsed.netloc:
|
||||||
|
if path.endswith(".gifv"):
|
||||||
|
return url.replace(".gifv", ".mp4"), "video"
|
||||||
|
return url, "image"
|
||||||
|
|
||||||
|
if "/a/" in path or "/gallery/" in path:
|
||||||
|
logger.debug(f"Imgur albums not supported: {url}")
|
||||||
|
return None, "image"
|
||||||
|
|
||||||
|
match = re.search(r"/(\w+)(?:\.\w+)?$", path)
|
||||||
|
if match:
|
||||||
|
image_id = match.group(1)
|
||||||
|
return f"https://i.imgur.com/{image_id}.jpg", "image"
|
||||||
|
|
||||||
|
return None, "image"
|
||||||
45
src/extractors/reddit.py
Normal file
45
src/extractors/reddit.py
Normal file
|
|
@ -0,0 +1,45 @@
|
||||||
|
"""Reddit video (v.redd.it) URL extractor using yt-dlp."""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from typing import Optional
|
||||||
|
|
||||||
|
logger = logging.getLogger("reddit_collector")
|
||||||
|
|
||||||
|
|
||||||
|
def extract_reddit_video_url(url: str) -> Optional[str]:
|
||||||
|
"""
|
||||||
|
Extract video URL from v.redd.it links.
|
||||||
|
Uses yt-dlp to get the actual video URL.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
import yt_dlp
|
||||||
|
|
||||||
|
ydl_opts = {
|
||||||
|
"quiet": True,
|
||||||
|
"no_warnings": True,
|
||||||
|
"extract_flat": False,
|
||||||
|
}
|
||||||
|
|
||||||
|
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||||||
|
info = ydl.extract_info(url, download=False)
|
||||||
|
|
||||||
|
if info and "url" in info:
|
||||||
|
return info["url"]
|
||||||
|
|
||||||
|
if info and "formats" in info:
|
||||||
|
mp4_formats = [
|
||||||
|
f for f in info["formats"]
|
||||||
|
if f.get("ext") == "mp4" and f.get("url")
|
||||||
|
]
|
||||||
|
if mp4_formats:
|
||||||
|
best = max(mp4_formats, key=lambda x: x.get("height", 0))
|
||||||
|
return best["url"]
|
||||||
|
|
||||||
|
return None
|
||||||
|
|
||||||
|
except ImportError:
|
||||||
|
logger.warning("yt-dlp not installed, cannot extract Reddit video URLs")
|
||||||
|
return None
|
||||||
|
except Exception as e:
|
||||||
|
logger.debug(f"Failed to extract Reddit video URL: {e}")
|
||||||
|
return None
|
||||||
Loading…
Add table
Add a link
Reference in a new issue