Compare commits
@@ -0,0 +1,28 @@
|
||||
name: Publish to Comfy registry
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
- master
|
||||
paths:
|
||||
- "pyproject.toml"
|
||||
|
||||
permissions:
|
||||
issues: write
|
||||
|
||||
jobs:
|
||||
publish-node:
|
||||
name: Publish Custom Node to registry
|
||||
runs-on: ubuntu-latest
|
||||
if: ${{ github.repository_owner == 'EricRollei' }}
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
submodules: true
|
||||
- name: Publish Custom Node
|
||||
uses: Comfy-Org/publish-node-action@v1
|
||||
with:
|
||||
## Add your own personal access token to your Github Repository secrets and reference it here.
|
||||
personal_access_token: ${{ secrets.REGISTRY_ACCESS_TOKEN }}
|
||||
+11
-3
@@ -117,7 +117,9 @@ configs/twitter_cookies.json
|
||||
configs/auth_config.json
|
||||
configs/gallery-dl.conf
|
||||
configs/gallery-dl-*.conf
|
||||
configs/yt-dlp.conf
|
||||
configs/web_scraper/auth_config.json
|
||||
configs/node_settings.json
|
||||
*.secret
|
||||
*.key
|
||||
|
||||
@@ -135,10 +137,13 @@ output/
|
||||
instagram-downloads/
|
||||
youtube-downloads/
|
||||
gallery-dl-downloads/
|
||||
gallery-dl-output/
|
||||
gallery-dl/
|
||||
web_scraper_output/
|
||||
test-downloads/
|
||||
temp/
|
||||
tmp/
|
||||
*.part
|
||||
|
||||
# Debug files
|
||||
debug/
|
||||
@@ -197,6 +202,9 @@ scratch/
|
||||
Docs/edit-docs-not-for-github/
|
||||
|
||||
# ============================================================================
|
||||
# Session files
|
||||
nodes/sessions/*.json
|
||||
|
||||
# Archives
|
||||
# ============================================================================
|
||||
|
||||
@@ -220,9 +228,9 @@ desktop.ini
|
||||
# Keep these files (force include)
|
||||
# ============================================================================
|
||||
|
||||
# Force include yt-dlp config templates (they're safe without credentials)
|
||||
!configs/yt-dlp.conf
|
||||
!configs/yt-dlp-*.conf
|
||||
# Force include yt-dlp config templates (safe presets without credentials)
|
||||
!configs/yt-dlp-hq.conf
|
||||
!configs/yt-dlp-audio.conf
|
||||
|
||||
# Force include all .example files for users to copy
|
||||
!configs/*.example
|
||||
|
||||
+32
-1
@@ -5,7 +5,38 @@ All notable changes to Download Tools for ComfyUI will be documented in this fil
|
||||
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
||||
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||
|
||||
## [Unreleased]
|
||||
## [0.8.5] - 2026-02-09
|
||||
|
||||
### Added
|
||||
- **BellazonHandler** — New IPS/Invision Community forum handler for bellazon.com
|
||||
- Full-resolution image extraction via `<a href>` links and `data-full-image` attributes
|
||||
- Zero-thumbnail guarantee with 3-layer `.thumb.` URL rejection (JS, Python, post_process)
|
||||
- Automatic multi-page pagination (detects total pages, navigates all)
|
||||
- Spoiler/hidden content auto-opening (`<details>` blocks, IPS spoiler markup)
|
||||
- YouTube/Vimeo video link collection with URL normalisation
|
||||
- Lightbox link support (`data-ipslightbox`)
|
||||
- Gallery-dl Downloader improvements:
|
||||
- Real-time progress output via threaded stdout/stderr streaming
|
||||
- ComfyUI cancel button support for long-running downloads
|
||||
- Skip already-organized files to avoid redundant processing
|
||||
- Seed parameter for forced re-execution
|
||||
- `configs/yt-dlp.conf.example` — Clean template for yt-dlp configuration
|
||||
|
||||
### Changed
|
||||
- Gallery-dl default timeout increased to 1800s (30 min), max to 36000s (10 hr)
|
||||
- README updated with comprehensive Web Image Scraper section and 25+ handler table
|
||||
- Updated config documentation to use `.example` pattern
|
||||
|
||||
### Fixed
|
||||
- IPS forum thumbnail URL leak — discovered that IPS uses different hashes for
|
||||
thumbnail vs full-res URLs; regex-stripping `.thumb.` produced wrong/404 URLs
|
||||
|
||||
### Security
|
||||
- Removed plaintext credentials from `configs/yt-dlp.conf` in repository
|
||||
- Added `configs/yt-dlp.conf` to `.gitignore` to prevent future credential exposure
|
||||
- Created `.example` template pattern for sensitive config files
|
||||
|
||||
## [0.8.3] - 2025-XX-XX
|
||||
|
||||
### Added
|
||||
- Initial public release preparation
|
||||
|
||||
@@ -6,15 +6,162 @@ ComfyUI custom nodes for downloading media from 1000+ websites including Instagr
|
||||
|
||||
- **Gallery-dl Node** - Download images and videos from 100+ websites
|
||||
- Instagram, Reddit, Twitter/X, DeviantArt, Pixiv, and more
|
||||
- Supports authentication via browser cookies
|
||||
- **Resolution filtering** - Skip small images (default: 768px minimum)
|
||||
- Supports authentication via config file or browser cookies
|
||||
- Automatic file organization
|
||||
- Download archive to avoid duplicates
|
||||
- **Persistent config paths** - Paths are remembered between sessions
|
||||
- **Configurable timeout** - Up to 1 hour for large downloads
|
||||
|
||||
- **Yt-dlp Node** - Download videos and audio from 1000+ platforms
|
||||
- **Web Image Scraper Node** - Extract images from any website via Playwright
|
||||
- 25+ site-specific handlers for optimised extraction
|
||||
- Automatic full-resolution image detection (skips thumbnails)
|
||||
- Multi-page pagination and infinite-scroll support
|
||||
- Spoiler / hidden-content auto-reveal (IPS forums)
|
||||
- Video link collection (YouTube / Vimeo) for use with yt-dlp
|
||||
- Duplicate detection via perceptual hashing
|
||||
- Configurable minimum dimensions, parallel downloads
|
||||
|
||||
- **Yt-dlp Node** - Download videos and audio from 1800+ platforms
|
||||
- YouTube, TikTok, Vimeo, Twitch, and more
|
||||
- Multiple quality options
|
||||
- Audio extraction support
|
||||
- Playlist support
|
||||
- Authentication via cookies or credentials
|
||||
|
||||
## 📺 Yt-dlp Supported Sites (Major Platforms)
|
||||
|
||||
The yt-dlp node supports **1800+ websites**. Here are the most popular ones:
|
||||
|
||||
### Video Platforms
|
||||
| Platform | Features | Auth Required |
|
||||
|----------|----------|---------------|
|
||||
| **YouTube** | Videos, playlists, channels, shorts, live streams, music | Optional (for age-restricted/private) |
|
||||
| **Vimeo** | Videos, albums, channels, on-demand, showcases | Optional (for private content) |
|
||||
| **Twitch** | VODs, clips, live streams | Optional |
|
||||
| **TikTok** | Videos, user profiles, collections | No |
|
||||
| **Dailymotion** | Videos, playlists, user content | Optional |
|
||||
| **Facebook** | Videos, reels, ads | No |
|
||||
| **Instagram** | Videos, reels, stories | No (use gallery-dl for images) |
|
||||
| **Twitter/X** | Videos, spaces, broadcasts | No |
|
||||
| **Reddit** | Video posts | No |
|
||||
| **Rumble** | Videos, channels | No |
|
||||
| **Kick** | VODs, clips, live streams | No |
|
||||
| **Odysee/LBRY** | Videos, channels, playlists | No |
|
||||
|
||||
### Streaming Services (May Require Subscription)
|
||||
| Platform | Notes |
|
||||
|----------|-------|
|
||||
| **Crunchyroll** | Anime streaming |
|
||||
| **Nebula** | Creator platform |
|
||||
| **CuriosityStream** | Documentaries |
|
||||
| **Dropout** | Comedy streaming |
|
||||
| **Patreon** | Creator videos |
|
||||
| **Floatplane** | Tech creator content |
|
||||
|
||||
### Music & Audio
|
||||
| Platform | Features |
|
||||
|----------|----------|
|
||||
| **SoundCloud** | Tracks, playlists, user content |
|
||||
| **Bandcamp** | Albums, tracks |
|
||||
| **Mixcloud** | Mixes, playlists |
|
||||
| **Audiomack** | Tracks, albums |
|
||||
| **Spotify** | Podcasts only (not music) |
|
||||
| **Apple Podcasts** | Podcast episodes |
|
||||
|
||||
### News & Media
|
||||
| Platform | Notes |
|
||||
|----------|-------|
|
||||
| **BBC iPlayer** | UK content |
|
||||
| **PBS** | US public broadcasting |
|
||||
| **CBS News** | News clips |
|
||||
| **NBC** | News and shows |
|
||||
| **CNN** | News clips |
|
||||
| **ESPN** | Sports clips |
|
||||
| **Arte** | European culture |
|
||||
|
||||
### Educational
|
||||
| Platform | Notes |
|
||||
|----------|-------|
|
||||
| **Khan Academy** | Free courses |
|
||||
| **TED** | Talks and playlists |
|
||||
| **Udemy** | Requires login |
|
||||
| **LinkedIn Learning** | Requires subscription |
|
||||
| **Coursera** | Some content |
|
||||
|
||||
### Other Popular Sites
|
||||
| Platform | Type |
|
||||
|----------|------|
|
||||
| **Bilibili** | Chinese video platform |
|
||||
| **NicoNico** | Japanese video platform |
|
||||
| **VK** | Russian social network |
|
||||
| **Weibo** | Chinese social media |
|
||||
| **Archive.org** | Internet Archive |
|
||||
| **Dropbox** | Shared videos |
|
||||
| **Google Drive** | Shared videos |
|
||||
| **Steam** | Game trailers |
|
||||
| **Imgur** | Video content |
|
||||
|
||||
### Full List
|
||||
For the complete list of 1800+ supported sites, see the [official yt-dlp supported sites](https://github.com/yt-dlp/yt-dlp/blob/master/supportedsites.md).
|
||||
|
||||
### Site-Specific Authentication
|
||||
|
||||
Some sites work better with authentication:
|
||||
|
||||
```bash
|
||||
# Using .netrc file (recommended for security)
|
||||
--netrc
|
||||
|
||||
# Using username/password directly
|
||||
--username YOUR_EMAIL --password YOUR_PASSWORD
|
||||
|
||||
# Using browser cookies
|
||||
--cookies-from-browser firefox
|
||||
```
|
||||
|
||||
**Tip:** For sites like Vimeo, YouTube (age-restricted), or any subscription service, authentication unlocks more content.
|
||||
|
||||
## 🌐 Web Image Scraper – Site Handlers
|
||||
|
||||
The Web Image Scraper node ships with **25+ site-specific handlers** that understand each site's DOM structure and extract full-resolution images automatically. A generic handler covers everything else.
|
||||
|
||||
| Handler | Sites | Key Features |
|
||||
|---------|-------|--------------|
|
||||
| **BellazonHandler** | bellazon.com (IPS / Invision Community forums) | Full-res from `<a>` hrefs (not thumbnails); auto-paginate all topic pages; opens spoiler / hidden-content blocks; collects YouTube & Vimeo video links |
|
||||
| **InstagramHandler** | instagram.com | Posts, reels, stories; cookie auth |
|
||||
| **RedditHandler** | reddit.com | Gallery posts, age-gate bypass |
|
||||
| **BskyHandler** | bsky.app | Bluesky AT Protocol API |
|
||||
| **PinterestHandler** | pinterest.com | Pin boards, infinite scroll |
|
||||
| **FlickrHandler** | flickr.com | Original-size downloads |
|
||||
| **DeviantArtHandler** | deviantart.com | Full-resolution deviations |
|
||||
| **ArtStationHandler** | artstation.com | Project galleries |
|
||||
| **BehanceHandler** | behance.net | Project modules |
|
||||
| **500pxHandler** | 500px.com | Photo pages |
|
||||
| **UnsplashHandler** | unsplash.com | Full-res downloads |
|
||||
| **TumblrHandler** | tumblr.com | Blog posts |
|
||||
| **KavyarHandler** | kavyar.com | Model portfolios |
|
||||
| **CosmosHandler** | cosmos.so | Boards & collections |
|
||||
| **GoogleArtsHandler** | artsandculture.google.com | High-res artwork tiles |
|
||||
| **ArtsyHandler** | artsy.net | Artwork pages |
|
||||
| **ModelMayhemHandler** | modelmayhem.com | Portfolio images |
|
||||
| **WixHandler** | Wix-powered sites | Wix media URLs |
|
||||
| **WordPressHandler** | WordPress sites | Featured images, galleries |
|
||||
| **YouTubeHandler** | youtube.com | Thumbnails, channel art |
|
||||
| **PortfolioHandler** | Generic portfolio sites | Common portfolio layouts |
|
||||
| **GenericHandler** | Any website | Fallback: extracts all images meeting size thresholds |
|
||||
|
||||
### IPS / Invision Community Forums (Bellazon)
|
||||
|
||||
The **BellazonHandler** is purpose-built for Invision Community (IPS) forums:
|
||||
|
||||
- **Full-resolution only** — IPS thumbnails use a different hash than the full-res image, so the handler reads the authoritative `<a href>` and `data-full-image` attributes instead of trying to rewrite thumbnail URLs.
|
||||
- **Spoiler handling** — Automatically opens `<details>` spoiler blocks ("Spoiler", "Spoiler Nudity", "Reveal hidden contents", etc.) before extraction so hidden images are included.
|
||||
- **Multi-page pagination** — Detects "PAGE X OF Y" controls and walks through every page of a topic automatically.
|
||||
- **Video link collection** — YouTube and Vimeo URLs embedded in posts are collected and returned as video items for optional download with the yt-dlp node.
|
||||
- **Zero thumbnail downloads** — Any URL containing `.thumb.` is hard-rejected at three levels (JS extraction, Python filter, post-processing).
|
||||
|
||||
To add support for another IPS-powered forum, just add its domain to the `IPS_DOMAINS` list in `site_handlers/bellazon_handler.py`.
|
||||
|
||||
## 📦 Installation
|
||||
|
||||
@@ -55,9 +202,11 @@ ComfyUI custom nodes for downloading media from 1000+ websites including Instagr
|
||||
1. Add "Gallery-dl Downloader" node to your workflow
|
||||
2. Enter a URL (e.g., Instagram profile, Reddit post)
|
||||
3. Configure options:
|
||||
- Enable `use_browser_cookies` for private content
|
||||
- Set `config_path` to `./configs/gallery-dl.conf` (auto-saved for next time)
|
||||
- Enable `filter_by_resolution` to skip small images (768px default)
|
||||
- Enable `organize_files` to sort by type
|
||||
- Enable `use_download_archive` to avoid duplicates
|
||||
- Increase `download_timeout` for large galleries (default: 600s)
|
||||
4. Execute!
|
||||
|
||||
**Supported Sites:** Instagram, Reddit, Twitter, DeviantArt, Pixiv, Tumblr, Pinterest, Flickr, and 90+ more. See [gallery-dl supported sites](https://github.com/mikf/gallery-dl/blob/master/docs/supportedsites.md).
|
||||
@@ -65,16 +214,65 @@ ComfyUI custom nodes for downloading media from 1000+ websites including Instagr
|
||||
### Yt-dlp Downloader
|
||||
|
||||
1. Add "Yt-dlp Downloader" node to your workflow
|
||||
2. Enter a URL (e.g., YouTube video)
|
||||
2. Enter a URL (e.g., YouTube video, Vimeo, Twitch VOD, TikTok)
|
||||
3. Choose format:
|
||||
- `best` - Best quality video
|
||||
- `best[height<=1080]` - Best quality up to 1080p
|
||||
- `audio-only` - Extract audio (requires FFmpeg)
|
||||
- Custom format string
|
||||
4. Execute!
|
||||
4. For private content, enable browser cookies or use config file with credentials
|
||||
5. Execute!
|
||||
|
||||
**Supported Sites:** YouTube, TikTok, Vimeo, Twitch, Facebook, Instagram, Twitter, and 1000+ more. See [yt-dlp supported sites](https://github.com/yt-dlp/yt-dlp/blob/master/supportedsites.md).
|
||||
**Supported Sites:** YouTube, Vimeo, TikTok, Twitch, Facebook, Instagram, Twitter, Dailymotion, SoundCloud, and 1800+ more. See [yt-dlp supported sites](https://github.com/yt-dlp/yt-dlp/blob/master/supportedsites.md) or the table above.
|
||||
|
||||
## 🔐 Authentication
|
||||
## � Instagram Downloads
|
||||
|
||||
**For Instagram, use Gallery-dl as it is better.** Gallery-dl has native Instagram API support, handles pagination, rate limits, and can download posts, stories, reels, and highlights.
|
||||
|
||||
### Recommended Settings
|
||||
|
||||
| Setting | Value | Notes |
|
||||
|---------|-------|-------|
|
||||
| `config_path` | `./configs/gallery-dl.conf` | Contains all site credentials (auto-saved) |
|
||||
| `cookie_file` | *(leave empty)* | Cookies are in config file |
|
||||
| `use_browser_cookies` | ❌ False | Config file is more reliable |
|
||||
| `filter_by_resolution` | ✅ True | Skip thumbnails and small images |
|
||||
| `min_image_width/height` | 768 | Minimum resolution in pixels |
|
||||
|
||||
### One Config For All Sites
|
||||
|
||||
The `gallery-dl.conf` file contains credentials for **all supported sites** (Instagram, Reddit, 500px, DeviantArt, Pinterest, Flickr, Bluesky). Gallery-dl automatically uses the right credentials based on the URL you're downloading from.
|
||||
|
||||
### Why Not Browser Cookies?
|
||||
|
||||
- **Chrome/Edge:** Require admin privileges AND browser must be closed
|
||||
- **Firefox:** Works without admin, but less reliable than config file
|
||||
- **Config file:** Most reliable method - always works
|
||||
|
||||
### Setting Up Instagram Authentication
|
||||
|
||||
1. **Export your Instagram cookies** from Chrome/Firefox using "Get cookies.txt LOCALLY" extension
|
||||
2. **Copy the key cookies** to `configs/gallery-dl.conf` in the `instagram` section:
|
||||
```json
|
||||
"instagram": {
|
||||
"cookies": {
|
||||
"sessionid": "YOUR_SESSION_ID",
|
||||
"ds_user_id": "YOUR_USER_ID",
|
||||
"csrftoken": "YOUR_CSRF_TOKEN",
|
||||
"mid": "YOUR_MID_VALUE"
|
||||
}
|
||||
}
|
||||
```
|
||||
3. **Use in node:**
|
||||
- Set `config_path` to `./configs/gallery-dl.conf`
|
||||
- Leave `cookie_file` empty
|
||||
- Set `use_browser_cookies` to False
|
||||
|
||||
### Cookie Expiration
|
||||
|
||||
Instagram session cookies expire after ~1 year. If downloads start failing with 401 errors, export fresh cookies from your browser.
|
||||
|
||||
## �🔐 Authentication
|
||||
|
||||
Many sites require authentication for private content:
|
||||
|
||||
@@ -97,13 +295,15 @@ Many sites require authentication for private content:
|
||||
|
||||
Config files are stored in `download-tools/configs/`:
|
||||
|
||||
- `gallery-dl.conf` - Gallery-dl settings
|
||||
- `gallery-dl-browser-cookies.conf` - Browser cookie configuration
|
||||
- `yt-dlp.conf` - Yt-dlp default settings
|
||||
- `gallery-dl.conf` - **Main config** with credentials for all sites (Instagram, Reddit, 500px, DeviantArt, Pinterest, Flickr, Bluesky)
|
||||
- `gallery-dl.conf.example` - Template for creating your own config
|
||||
- `yt-dlp.conf.example` - Yt-dlp template (copy to `yt-dlp.conf` and add your credentials)
|
||||
- `yt-dlp-audio.conf` - Audio extraction preset
|
||||
- `yt-dlp-hq.conf` - High quality preset
|
||||
|
||||
You can create custom config files and reference them in the nodes.
|
||||
### Persistent Paths
|
||||
|
||||
Config file paths are **automatically saved** between ComfyUI sessions. Once you set `config_path` or `cookie_file`, they'll be remembered for next time.
|
||||
|
||||
## 📚 Documentation
|
||||
|
||||
@@ -120,15 +320,29 @@ Detailed guides available in `Docs/`:
|
||||
Already installed! It's a Python package. The node will find it automatically.
|
||||
|
||||
### "Chrome cookies not accessible"
|
||||
Try:
|
||||
Chrome locks its cookie database while running. Solutions:
|
||||
- Close Chrome completely before running
|
||||
- Run ComfyUI as administrator
|
||||
- Use Firefox instead
|
||||
- Export cookies manually (Method 2)
|
||||
- Use Firefox instead (doesn't require admin)
|
||||
- **Best:** Use config file with cookies (see Instagram section above)
|
||||
|
||||
### "Instagram/Reddit downloads fail"
|
||||
Authentication required:
|
||||
- Enable `use_browser_cookies`
|
||||
- Or export cookies from logged-in browser
|
||||
- Use `config_path: ./configs/gallery-dl.conf` with cookies
|
||||
- Or enable `use_browser_cookies` with Firefox
|
||||
- 401 Unauthorized = expired cookies, export fresh ones
|
||||
|
||||
### "Downloads timing out"
|
||||
For large galleries:
|
||||
- Increase `download_timeout` (default: 1800s / 30 min, max: 10 hours)
|
||||
- Large Instagram profiles may need several hours
|
||||
- If a download times out, just run again — gallery-dl automatically resumes using the download archive
|
||||
|
||||
### "Too many small images"
|
||||
Enable resolution filtering:
|
||||
- Set `filter_by_resolution` to True
|
||||
- Adjust `min_image_width` and `min_image_height` (default: 768px)
|
||||
- Videos are never filtered, only images
|
||||
|
||||
### "CUDA out of memory" / "FFmpeg not found"
|
||||
For audio extraction:
|
||||
|
||||
+15
-18
@@ -27,6 +27,7 @@ See CREDITS.md for complete list of dependencies and their licenses.
|
||||
|
||||
import os
|
||||
import sys
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
|
||||
# Initialize node mappings
|
||||
@@ -56,30 +57,26 @@ def load_nodes():
|
||||
if filename.endswith(".py") and not filename.startswith("_"):
|
||||
module_name = filename[:-3]
|
||||
try:
|
||||
# Import the module
|
||||
# Import the module using importlib
|
||||
module_path = os.path.join(nodes_dir, filename)
|
||||
print(f"[Download Tools] Loading node: {module_name}")
|
||||
|
||||
# Read and execute the module
|
||||
with open(module_path, 'r', encoding='utf-8') as f:
|
||||
module_code = f.read()
|
||||
|
||||
# Create a module namespace
|
||||
module_namespace = {
|
||||
'__name__': f'download_tools.nodes.{module_name}',
|
||||
'__file__': module_path,
|
||||
}
|
||||
|
||||
# Execute the module code
|
||||
exec(module_code, module_namespace)
|
||||
# Use importlib for safe module loading
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
f'download_tools.nodes.{module_name}',
|
||||
module_path
|
||||
)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
sys.modules[spec.name] = module
|
||||
spec.loader.exec_module(module)
|
||||
|
||||
# Extract NODE_CLASS_MAPPINGS and NODE_DISPLAY_NAME_MAPPINGS
|
||||
if 'NODE_CLASS_MAPPINGS' in module_namespace:
|
||||
NODE_CLASS_MAPPINGS.update(module_namespace['NODE_CLASS_MAPPINGS'])
|
||||
print(f"[Download Tools] ✓ Registered classes from {module_name}: {list(module_namespace['NODE_CLASS_MAPPINGS'].keys())}")
|
||||
if hasattr(module, 'NODE_CLASS_MAPPINGS'):
|
||||
NODE_CLASS_MAPPINGS.update(module.NODE_CLASS_MAPPINGS)
|
||||
print(f"[Download Tools] ✓ Registered classes from {module_name}: {list(module.NODE_CLASS_MAPPINGS.keys())}")
|
||||
|
||||
if 'NODE_DISPLAY_NAME_MAPPINGS' in module_namespace:
|
||||
NODE_DISPLAY_NAME_MAPPINGS.update(module_namespace['NODE_DISPLAY_NAME_MAPPINGS'])
|
||||
if hasattr(module, 'NODE_DISPLAY_NAME_MAPPINGS'):
|
||||
NODE_DISPLAY_NAME_MAPPINGS.update(module.NODE_DISPLAY_NAME_MAPPINGS)
|
||||
|
||||
except Exception as e:
|
||||
print(f"[Download Tools] Error loading {module_name}: {e}")
|
||||
|
||||
@@ -38,6 +38,16 @@
|
||||
"domain": "example.com",
|
||||
"username": "YOUR_USERNAME",
|
||||
"password": "YOUR_PASSWORD"
|
||||
},
|
||||
"modelmayhem.com": {
|
||||
"auth_type": "form",
|
||||
"domain": "modelmayhem.com",
|
||||
"username": "YOUR_EMAIL",
|
||||
"password": "YOUR_PASSWORD",
|
||||
"timeout": 30000,
|
||||
"scroll_delay_ms": 1500,
|
||||
"max_scroll_count": 15,
|
||||
"note": "Login credentials for ModelMayhem portfolio access"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -9,7 +9,21 @@
|
||||
"client-secret": "YOUR_REDDIT_CLIENT_SECRET",
|
||||
"user-agent": "MediaTool (by /u/YOUR_USERNAME)",
|
||||
"username": "YOUR_REDDIT_USERNAME",
|
||||
"password": "YOUR_REDDIT_PASSWORD"
|
||||
"password": "YOUR_REDDIT_PASSWORD",
|
||||
"comments": 0,
|
||||
"videos": true
|
||||
},
|
||||
|
||||
"redgifs": {
|
||||
"format": ["hd", "sd", "gif"]
|
||||
},
|
||||
|
||||
"gfycat": {
|
||||
"format": ["mp4", "webm", "gif"]
|
||||
},
|
||||
|
||||
"imgur": {
|
||||
"mp4": true
|
||||
},
|
||||
|
||||
"500px": {
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
# yt-dlp configuration file
|
||||
# Basic settings for most downloads
|
||||
# Copy this file to yt-dlp.conf and customise with your own settings.
|
||||
# yt-dlp.conf is gitignored so your credentials stay local.
|
||||
|
||||
# Output filename template
|
||||
--output "%(uploader)s/%(title)s.%(ext)s"
|
||||
@@ -23,8 +24,13 @@
|
||||
--sleep-interval 1
|
||||
--max-sleep-interval 5
|
||||
|
||||
# Cookie settings
|
||||
# Cookie settings (pick one method)
|
||||
--cookies-from-browser firefox
|
||||
# --cookies /path/to/cookies.txt
|
||||
|
||||
# Authentication (uncomment and fill in for sites that require login)
|
||||
# --username YOUR_EMAIL
|
||||
# --password YOUR_PASSWORD
|
||||
|
||||
# Subtitle settings
|
||||
--write-subs
|
||||
+347
-51
@@ -48,10 +48,31 @@ import json
|
||||
import subprocess
|
||||
import tempfile
|
||||
import urllib.parse
|
||||
import folder_paths
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Tuple
|
||||
|
||||
# Import persistent settings manager
|
||||
try:
|
||||
from ..utils.persistent_settings import get_persistent_setting, set_persistent_setting
|
||||
PERSISTENT_SETTINGS_AVAILABLE = True
|
||||
except ImportError:
|
||||
try:
|
||||
import sys
|
||||
utils_dir = os.path.dirname(os.path.dirname(__file__))
|
||||
if utils_dir not in sys.path:
|
||||
sys.path.insert(0, utils_dir)
|
||||
from utils.persistent_settings import get_persistent_setting, set_persistent_setting
|
||||
PERSISTENT_SETTINGS_AVAILABLE = True
|
||||
except ImportError:
|
||||
PERSISTENT_SETTINGS_AVAILABLE = False
|
||||
def get_persistent_setting(node_type, key, default=""):
|
||||
return default
|
||||
def set_persistent_setting(node_type, key, value):
|
||||
return False
|
||||
print("Warning: persistent_settings not available. Config paths will not persist.")
|
||||
|
||||
|
||||
class GalleryDLDownloader:
|
||||
def __init__(
|
||||
@@ -71,6 +92,12 @@ class GalleryDLDownloader:
|
||||
# New advanced options
|
||||
instagram_include: str = "posts",
|
||||
extra_options: str = "",
|
||||
# Minimum resolution filtering
|
||||
min_image_width: int = 768,
|
||||
min_image_height: int = 768,
|
||||
filter_by_resolution: bool = True,
|
||||
# Timeout settings
|
||||
download_timeout: int = 1800,
|
||||
):
|
||||
self.url_list = url_list or []
|
||||
self.url_file = url_file
|
||||
@@ -86,6 +113,10 @@ class GalleryDLDownloader:
|
||||
self.organize_files = organize_files
|
||||
self.instagram_include = instagram_include
|
||||
self.extra_options = extra_options
|
||||
self.min_image_width = min_image_width
|
||||
self.min_image_height = min_image_height
|
||||
self.filter_by_resolution = filter_by_resolution
|
||||
self.download_timeout = download_timeout
|
||||
|
||||
# Initialize debug info
|
||||
self.debug_info = [] # Store debug information for status reporting
|
||||
@@ -105,11 +136,6 @@ class GalleryDLDownloader:
|
||||
|
||||
command += ["-d", self.output_dir]
|
||||
|
||||
# Use config file if provided
|
||||
if self.config_path:
|
||||
command += ["--config", self.config_path]
|
||||
self.debug_info.append(f"📄 Using config file: {self.config_path}")
|
||||
|
||||
# Detect target sites for better authentication handling
|
||||
target_sites = set()
|
||||
for url in urls:
|
||||
@@ -119,7 +145,20 @@ class GalleryDLDownloader:
|
||||
target_sites.add('reddit')
|
||||
elif 'twitter.com' in url or 'x.com' in url:
|
||||
target_sites.add('twitter')
|
||||
# Add more site detection as needed
|
||||
elif 'flickr.com' in url:
|
||||
target_sites.add('flickr')
|
||||
elif 'deviantart.com' in url:
|
||||
target_sites.add('deviantart')
|
||||
elif '500px.com' in url:
|
||||
target_sites.add('500px')
|
||||
elif 'pinterest' in url:
|
||||
target_sites.add('pinterest')
|
||||
elif 'bsky.app' in url or 'bluesky' in url:
|
||||
target_sites.add('bluesky')
|
||||
elif 'tumblr.com' in url:
|
||||
target_sites.add('tumblr')
|
||||
elif 'artstation.com' in url:
|
||||
target_sites.add('artstation')
|
||||
|
||||
if target_sites:
|
||||
self.debug_info.append(f"🎯 Detected target sites: {', '.join(target_sites)}")
|
||||
@@ -129,23 +168,30 @@ class GalleryDLDownloader:
|
||||
self.debug_info.append("⚠️ WARNING: Reddit may hang or require updated API credentials")
|
||||
self.debug_info.append(" Consider testing with Instagram or other sites first")
|
||||
|
||||
# Handle cookie file (takes precedence over browser cookies)
|
||||
if self.cookie_file:
|
||||
# Warn if using cookies for non-matching sites (but allow Instagram + config combo)
|
||||
if target_sites and 'instagram' not in target_sites:
|
||||
self.debug_info.append(f"⚠️ Warning: Using Instagram cookies for {', '.join(target_sites)} - this may cause authentication conflicts")
|
||||
elif target_sites and 'instagram' in target_sites:
|
||||
self.debug_info.append("🔗 Using Instagram cookies - optimal setup!")
|
||||
|
||||
# Use config file if provided (contains site-specific credentials)
|
||||
if self.config_path:
|
||||
command += ["--config", self.config_path]
|
||||
self.debug_info.append(f"📄 Using config file: {self.config_path}")
|
||||
|
||||
# Handle authentication: Browser cookies toggle vs cookie file
|
||||
# When use_browser_cookies is TRUE: use browser cookies (even if cookie file exists)
|
||||
# When use_browser_cookies is FALSE: use cookie file if provided
|
||||
if self.use_browser_cookies:
|
||||
# User explicitly wants browser cookies
|
||||
command += ["--cookies-from-browser", self.browser_name]
|
||||
self.debug_info.append(f"🍪 Using cookies from {self.browser_name} browser (toggle enabled)")
|
||||
if self.cookie_file:
|
||||
self.debug_info.append(f" ℹ️ Note: Cookie file '{self.cookie_file}' ignored - browser cookies selected")
|
||||
# Test if browser cookies can be accessed
|
||||
self._test_browser_cookie_access()
|
||||
elif self.cookie_file:
|
||||
# Use cookie file when browser cookies toggle is off
|
||||
converted_cookie_file = self._convert_cookie_file()
|
||||
if converted_cookie_file:
|
||||
command += ["--cookies", converted_cookie_file]
|
||||
self.debug_info.append(f"🍪 Using cookie file: {converted_cookie_file}")
|
||||
elif self.use_browser_cookies:
|
||||
command += ["--cookies-from-browser", self.browser_name]
|
||||
self.debug_info.append(f"🍪 Extracting cookies from {self.browser_name} browser")
|
||||
# Test if browser cookies can be accessed
|
||||
self._test_browser_cookie_access()
|
||||
if 'instagram' in target_sites:
|
||||
self.debug_info.append("🔗 Instagram + cookie file - optimal setup!")
|
||||
|
||||
if self.use_download_archive:
|
||||
command += ["--download-archive", self.archive_file]
|
||||
@@ -158,6 +204,39 @@ class GalleryDLDownloader:
|
||||
]
|
||||
self.debug_info.append("🎬 Video files will be skipped")
|
||||
|
||||
# Add minimum resolution filter for images if enabled
|
||||
if self.filter_by_resolution and (self.min_image_width > 0 or self.min_image_height > 0):
|
||||
# gallery-dl filter expression for minimum resolution
|
||||
# IMPORTANT: Not all extractors provide width/height metadata (e.g., Bunkr, Cyberdrop)
|
||||
# gallery-dl uses eval() with metadata as locals(), so we can use locals().get()
|
||||
# If width/height don't exist, we default to 99999 so the file passes the filter
|
||||
# (download files with unknown dimensions rather than skip them)
|
||||
filter_expr = []
|
||||
if self.min_image_width > 0:
|
||||
# If width is not available in metadata, default to 99999 (pass filter)
|
||||
filter_expr.append(f"locals().get('width', 99999) >= {self.min_image_width}")
|
||||
if self.min_image_height > 0:
|
||||
# If height is not available in metadata, default to 99999 (pass filter)
|
||||
filter_expr.append(f"locals().get('height', 99999) >= {self.min_image_height}")
|
||||
|
||||
# Combine with video filter if needed, or apply image filter
|
||||
# Note: gallery-dl only allows one --filter, so we need to combine
|
||||
if self.skip_videos:
|
||||
# Already added video filter above, need to modify it
|
||||
# Remove the last filter we added and combine
|
||||
command = [c for c in command if c != "--filter"]
|
||||
command = [c for c in command if "extension not in" not in c]
|
||||
combined_filter = f"(extension not in ('mp4', 'webm', 'mkv', 'avi', 'mov', 'wmv', 'flv', 'm4v')) and ({' and '.join(filter_expr)})"
|
||||
command += ["--filter", combined_filter]
|
||||
else:
|
||||
# Only filter images by resolution (videos pass through)
|
||||
# Apply filter only to images, let videos through
|
||||
# Also pass through if width/height are not available (unknown dimensions)
|
||||
filter_condition = f"(extension in ('mp4', 'webm', 'mkv', 'avi', 'mov', 'wmv', 'flv', 'm4v')) or ({' and '.join(filter_expr)})"
|
||||
command += ["--filter", filter_condition]
|
||||
|
||||
self.debug_info.append(f"📐 Minimum resolution filter: {self.min_image_width}x{self.min_image_height}px (files with unknown dimensions will be downloaded)")
|
||||
|
||||
# Add rate limiting to be respectful to servers (use correct format)
|
||||
command += ["--sleep", "1.0"]
|
||||
|
||||
@@ -239,7 +318,8 @@ class GalleryDLDownloader:
|
||||
return []
|
||||
|
||||
def _organize_files_by_type(self, downloaded_files: List[str]) -> None:
|
||||
"""Sort downloaded files into subfolders by type (images, videos, etc.) within each profile directory"""
|
||||
"""Sort downloaded files into subfolders by type (images, videos, etc.) within each profile directory.
|
||||
Only organizes files that are NOT already in images/, videos/, audio/, or other/ folders."""
|
||||
if not downloaded_files:
|
||||
self.debug_info.append("📂 No files to organize")
|
||||
return
|
||||
@@ -250,6 +330,9 @@ class GalleryDLDownloader:
|
||||
video_extensions = {'.mp4', '.webm', '.mkv', '.avi', '.mov', '.wmv', '.flv', '.m4v', '.3gp', '.ogv', '.mpg', '.mpeg'}
|
||||
audio_extensions = {'.mp3', '.wav', '.flac', '.aac', '.ogg', '.m4a', '.wma', '.opus'}
|
||||
|
||||
# Folders that indicate file is already organized - skip these
|
||||
organized_folder_names = {'images', 'videos', 'audio', 'other'}
|
||||
|
||||
# Group files by their parent directory (profile/site directories)
|
||||
files_by_profile = {}
|
||||
|
||||
@@ -257,6 +340,12 @@ class GalleryDLDownloader:
|
||||
if not os.path.exists(file_path):
|
||||
self.debug_info.append(f"⚠️ File no longer exists: {os.path.basename(file_path)}")
|
||||
continue
|
||||
|
||||
# Skip files that are already in an organized folder
|
||||
parent_folder = os.path.basename(os.path.dirname(file_path))
|
||||
if parent_folder.lower() in organized_folder_names:
|
||||
self.debug_info.append(f"⏭️ Skipping already-organized: {os.path.basename(file_path)}")
|
||||
continue
|
||||
|
||||
# Skip files that shouldn't be organized
|
||||
filename = os.path.basename(file_path)
|
||||
@@ -416,6 +505,8 @@ class GalleryDLDownloader:
|
||||
|
||||
# Get list of files before download to track new files
|
||||
existing_files = set()
|
||||
# Also track filenames in organized folders to avoid re-organizing
|
||||
organized_folder_names = {'images', 'videos', 'audio', 'other'}
|
||||
if os.path.exists(self.output_dir):
|
||||
for root, dirs, files in os.walk(self.output_dir):
|
||||
for file in files:
|
||||
@@ -425,13 +516,27 @@ class GalleryDLDownloader:
|
||||
# Build and execute command
|
||||
command = self._build_command(urls)
|
||||
|
||||
# Execute download process with timeout handling
|
||||
# Execute download process with timeout handling and real-time output
|
||||
process_success = True
|
||||
process_returncode = 0
|
||||
stdout = ""
|
||||
stderr = ""
|
||||
stdout_lines = []
|
||||
stderr_lines = []
|
||||
|
||||
# Import ComfyUI's interrupt mechanism
|
||||
try:
|
||||
from comfy.model_management import processing_interrupted
|
||||
except ImportError:
|
||||
processing_interrupted = lambda: False
|
||||
|
||||
try:
|
||||
# Print command info for console logging
|
||||
print(f"\n{'='*60}")
|
||||
print(f"🚀 Gallery-dl Download Starting")
|
||||
print(f"📥 URLs: {len(urls)}")
|
||||
print(f"⏱️ Timeout: {self.download_timeout}s ({self.download_timeout // 60} min)")
|
||||
print(f"🛑 To cancel: Press Cancel button in ComfyUI")
|
||||
print(f"{'='*60}\n")
|
||||
|
||||
process = subprocess.Popen(
|
||||
command,
|
||||
stdout=subprocess.PIPE,
|
||||
@@ -439,27 +544,126 @@ class GalleryDLDownloader:
|
||||
universal_newlines=True,
|
||||
cwd=self.output_dir, # Set working directory
|
||||
)
|
||||
stdout, stderr = process.communicate(timeout=300) # 5 minute timeout
|
||||
process_returncode = process.returncode
|
||||
|
||||
# Use threading to read output without blocking
|
||||
import time
|
||||
import threading
|
||||
import queue
|
||||
|
||||
output_queue = queue.Queue()
|
||||
|
||||
def reader_thread(pipe, q):
|
||||
try:
|
||||
for line in iter(pipe.readline, ''):
|
||||
q.put(line)
|
||||
pipe.close()
|
||||
except:
|
||||
pass
|
||||
|
||||
# Start reader thread
|
||||
reader = threading.Thread(target=reader_thread, args=(process.stdout, output_queue))
|
||||
reader.daemon = True
|
||||
reader.start()
|
||||
|
||||
start_time = time.time()
|
||||
file_count = 0
|
||||
cancelled = False
|
||||
|
||||
while True:
|
||||
# Check for ComfyUI cancel button FIRST
|
||||
if processing_interrupted():
|
||||
process.kill()
|
||||
process.wait()
|
||||
cancelled = True
|
||||
process_success = False
|
||||
process_returncode = -2
|
||||
stderr_lines.append("Download cancelled by user")
|
||||
self.debug_info.append(f"🛑 Download cancelled by user after {file_count} files")
|
||||
print(f"\n🛑 CANCELLED by user - {file_count} files downloaded")
|
||||
print(f"💡 TIP: Run again to continue - already-downloaded files will be skipped!")
|
||||
break
|
||||
|
||||
# Check timeout
|
||||
elapsed = time.time() - start_time
|
||||
if elapsed > self.download_timeout:
|
||||
process.kill()
|
||||
process.wait()
|
||||
process_success = False
|
||||
process_returncode = -1
|
||||
stderr_lines.append(f"Download process timed out after {self.download_timeout // 60} minutes")
|
||||
self.debug_info.append(f"⏰ Download timed out after {self.download_timeout}s ({file_count} files downloaded)")
|
||||
print(f"\n⏰ TIMEOUT after {elapsed:.0f}s - {file_count} files downloaded")
|
||||
print(f"💡 TIP: Run again to continue - already-downloaded files will be skipped automatically!")
|
||||
break
|
||||
|
||||
# Check if process has finished
|
||||
if process.poll() is not None:
|
||||
break
|
||||
|
||||
# Read available output from queue (non-blocking)
|
||||
try:
|
||||
while True:
|
||||
try:
|
||||
line = output_queue.get_nowait()
|
||||
if line:
|
||||
stdout_lines.append(line.rstrip())
|
||||
# Print progress to console
|
||||
if "# " in line or line.strip().endswith(('.jpg', '.png', '.gif', '.mp4', '.webp', '.webm')):
|
||||
file_count += 1
|
||||
# Print every 10 files or specific milestones
|
||||
if file_count % 10 == 0 or file_count in [1, 5, 25, 50, 100]:
|
||||
print(f"📥 Downloaded: {file_count} files ({elapsed:.0f}s elapsed)")
|
||||
except queue.Empty:
|
||||
break
|
||||
except:
|
||||
pass
|
||||
|
||||
# Sleep 1 second between cancel checks to reduce CPU usage
|
||||
time.sleep(1.0)
|
||||
|
||||
# Get any remaining output
|
||||
if not cancelled:
|
||||
try:
|
||||
remaining_stdout, remaining_stderr = process.communicate(timeout=5)
|
||||
if remaining_stdout:
|
||||
stdout_lines.extend(remaining_stdout.splitlines())
|
||||
if remaining_stderr:
|
||||
stderr_lines.extend(remaining_stderr.splitlines())
|
||||
except:
|
||||
pass
|
||||
|
||||
process_returncode = process.returncode if process.returncode is not None else 0
|
||||
|
||||
if not cancelled and process_returncode != -1:
|
||||
print(f"\n{'='*60}")
|
||||
print(f"✅ Download completed: {file_count} files in {time.time() - start_time:.0f}s")
|
||||
print(f"{'='*60}\n")
|
||||
|
||||
except subprocess.TimeoutExpired:
|
||||
process.kill()
|
||||
process_success = False
|
||||
process_returncode = -1
|
||||
stderr = "Download process timed out after 5 minutes"
|
||||
stdout = ""
|
||||
self.debug_info.append("⏰ Download timed out, but continuing with file organization...")
|
||||
stderr_lines.append(f"Download process timed out after {self.download_timeout // 60} minutes")
|
||||
self.debug_info.append(f"⏰ Download timed out after {self.download_timeout}s, but continuing with file organization...")
|
||||
except Exception as e:
|
||||
process_success = False
|
||||
process_returncode = -1
|
||||
stderr = f"Download process failed: {str(e)}"
|
||||
stdout = ""
|
||||
stderr_lines.append(f"Download process failed: {str(e)}")
|
||||
self.debug_info.append(f"❌ Download failed: {e}, but continuing with file organization...")
|
||||
|
||||
stdout = "\n".join(stdout_lines)
|
||||
stderr = "\n".join(stderr_lines)
|
||||
|
||||
# Count only newly downloaded files (files that weren't there before)
|
||||
new_downloaded_files = []
|
||||
all_current_files = set()
|
||||
if os.path.exists(self.output_dir):
|
||||
for root, dirs, files in os.walk(self.output_dir):
|
||||
# Skip scanning inside organized folders - those files are already handled
|
||||
parent_folder = os.path.basename(root)
|
||||
if parent_folder.lower() in organized_folder_names:
|
||||
continue
|
||||
|
||||
for file in files:
|
||||
# Skip metadata, config, and cookie files more comprehensively
|
||||
if file.endswith((".json", ".txt", ".log", ".tmp")) or file.startswith("tmp") or "cookie" in file.lower():
|
||||
@@ -670,65 +874,114 @@ class GalleryDLNode:
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
# Load saved config paths from persistent settings
|
||||
saved_config_path = get_persistent_setting("gallery_dl", "config_path", "")
|
||||
saved_cookie_file = get_persistent_setting("gallery_dl", "cookie_file", "")
|
||||
|
||||
return {
|
||||
"required": {
|
||||
"url_list": ("STRING", {
|
||||
"multiline": True,
|
||||
"default": "# Enter URLs here, one per line\n# Example:\n# https://imgur.com/gallery/example\n# https://example.com/image.jpg"
|
||||
"default": "# Enter URLs here, one per line\n# Supports 100+ sites: Instagram, Reddit, Twitter/X, DeviantArt, Pixiv, 500px, Flickr, Pinterest, Bluesky, and more\n# Examples:\n# https://www.instagram.com/username/\n# https://www.reddit.com/r/subreddit/\n# https://twitter.com/username",
|
||||
"tooltip": "Enter URLs to download from, one per line. Supports Instagram, Reddit, Twitter/X, DeviantArt, Pixiv, 500px, Flickr, Pinterest, Bluesky, and 90+ other sites. Lines starting with # are ignored."
|
||||
}),
|
||||
"output_dir": ("STRING", {
|
||||
"default": "./gallery-dl-output",
|
||||
"tooltip": "Directory where downloaded files will be saved"
|
||||
"tooltip": "Directory where downloaded files will be saved. Relative paths are relative to ComfyUI's output folder. Files are organized into subfolders by site and user."
|
||||
})
|
||||
},
|
||||
"optional": {
|
||||
"url_file": ("STRING", {
|
||||
"default": "",
|
||||
"tooltip": "Path to a text file containing URLs (one per line)"
|
||||
"tooltip": "Optional: Path to a text file containing URLs (one per line). Useful for batch downloading from a prepared list. Leave empty to use url_list above."
|
||||
}),
|
||||
"config_path": ("STRING", {
|
||||
"default": "",
|
||||
"tooltip": "Path to gallery-dl config file. For Instagram: LEAVE EMPTY (not needed). For Reddit: use './configs/gallery-dl-no-reddit.conf' to disable Reddit or './configs/gallery-dl.conf' with your API credentials."
|
||||
"default": saved_config_path,
|
||||
"tooltip": "Path to gallery-dl.conf config file. This single file contains credentials for ALL sites (Instagram, Reddit, 500px, etc.) - gallery-dl auto-selects the right ones based on URL. Path is auto-saved for next time. Recommended: ./configs/gallery-dl.conf"
|
||||
}),
|
||||
"cookie_file": ("STRING", {
|
||||
"default": "",
|
||||
"tooltip": "Path to exported cookie JSON file. For Instagram: use './configs/instagram_cookies.json'. For Reddit: LEAVE EMPTY (use browser cookies instead)."
|
||||
"default": saved_cookie_file,
|
||||
"tooltip": "Path to exported cookie file (Netscape/JSON format). Only used when 'Use Browser Cookies' is OFF and you need site-specific cookies not in config. Usually leave empty - config file credentials work better. Path is auto-saved."
|
||||
}),
|
||||
"use_browser_cookies": ("BOOLEAN", {
|
||||
"default": True,
|
||||
"tooltip": "Use browser cookies for authentication (ignored if cookie file is provided)"
|
||||
"default": False,
|
||||
"label_on": "Browser Cookies",
|
||||
"label_off": "Config/Cookie File",
|
||||
"tooltip": "OFF (recommended): Use credentials from config file - most reliable. ON: Extract cookies directly from browser (Firefox works best; Chrome/Edge require admin rights and browser closed)."
|
||||
}),
|
||||
"browser_name": (["firefox", "chrome", "chromium", "edge", "safari", "opera"], {
|
||||
"default": "firefox",
|
||||
"tooltip": "Browser to extract cookies from (Firefox works without admin, Chrome/Edge require admin)"
|
||||
"tooltip": "Browser to extract cookies from when 'Use Browser Cookies' is ON. Firefox recommended - works without admin rights. Chrome/Edge need admin + browser closed."
|
||||
}),
|
||||
"use_download_archive": ("BOOLEAN", {
|
||||
"default": True,
|
||||
"tooltip": "Use archive file to skip already downloaded content"
|
||||
"label_on": "Skip Downloaded",
|
||||
"label_off": "Re-download All",
|
||||
"tooltip": "ON (recommended): Track downloaded files in SQLite database to skip duplicates on future runs. OFF: Re-download everything, may create duplicates."
|
||||
}),
|
||||
"archive_file": ("STRING", {
|
||||
"default": "./gallery-dl-archive.sqlite3",
|
||||
"tooltip": "Path to the download archive database"
|
||||
"tooltip": "Path to download archive database (SQLite). Tracks what's been downloaded to avoid duplicates. Delete this file to force re-downloading everything."
|
||||
}),
|
||||
"skip_videos": ("BOOLEAN", {
|
||||
"default": False,
|
||||
"tooltip": "Skip video files, download only images"
|
||||
"label_on": "Images Only",
|
||||
"label_off": "Images + Videos",
|
||||
"tooltip": "ON: Download only images, skip all video files. OFF: Download both images and videos. Useful when you only want still images from a mixed gallery."
|
||||
}),
|
||||
"filter_by_resolution": ("BOOLEAN", {
|
||||
"default": True,
|
||||
"label_on": "Filter Small Images",
|
||||
"label_off": "Keep All Sizes",
|
||||
"tooltip": "ON: Skip images smaller than min_image_width/height (removes thumbnails, icons, low-res images). Videos are never filtered. Sites without dimension metadata (Bunkr, Cyberdrop) will download all images. OFF: Download all images regardless of size."
|
||||
}),
|
||||
"min_image_width": ("INT", {
|
||||
"default": 768,
|
||||
"min": 0,
|
||||
"max": 8192,
|
||||
"step": 64,
|
||||
"tooltip": "Minimum image width in pixels when filter_by_resolution is ON. Images narrower than this are skipped. Set to 0 to disable width filtering. Note: Sites that don't provide dimension metadata will download all images."
|
||||
}),
|
||||
"min_image_height": ("INT", {
|
||||
"default": 768,
|
||||
"min": 0,
|
||||
"max": 8192,
|
||||
"step": 64,
|
||||
"tooltip": "Minimum image height in pixels when filter_by_resolution is ON. Images shorter than this are skipped. Set to 0 to disable height filtering. Note: Sites that don't provide dimension metadata will download all images."
|
||||
}),
|
||||
"extract_metadata": ("BOOLEAN", {
|
||||
"default": True,
|
||||
"tooltip": "Save download metadata to JSON file"
|
||||
"label_on": "Save Metadata",
|
||||
"label_off": "No Metadata",
|
||||
"tooltip": "ON: Save download metadata to JSON file (URLs, timestamps, file counts, errors). Useful for tracking what was downloaded. OFF: No metadata file created."
|
||||
}),
|
||||
"organize_files": ("BOOLEAN", {
|
||||
"default": True,
|
||||
"tooltip": "Sort downloaded files into subfolders (images/, videos/, audio/, other/)"
|
||||
"label_on": "Sort by Type",
|
||||
"label_off": "All in One Folder",
|
||||
"tooltip": "ON: Sort downloaded files into subfolders by type (images/, videos/, audio/, other/). OFF: Put all files in the main output folder without sorting."
|
||||
}),
|
||||
"instagram_include": (["posts", "stories", "highlights", "reels", "tagged", "info", "avatar", "all", "posts,stories", "posts,reels", "stories,highlights"], {
|
||||
"default": "posts",
|
||||
"tooltip": "For Instagram profiles: What to include (posts, stories, highlights, reels, tagged, info, avatar, or all)"
|
||||
"tooltip": "Instagram only: What content to download. 'posts' = feed posts, 'stories' = current stories (24hr), 'highlights' = saved story highlights, 'reels' = reels videos, 'all' = everything. Combine with comma: 'posts,reels'."
|
||||
}),
|
||||
"download_timeout": ("INT", {
|
||||
"default": 1800,
|
||||
"min": 60,
|
||||
"max": 36000,
|
||||
"step": 60,
|
||||
"tooltip": "Maximum time in seconds for this run. Default 1800s (30 min). For huge galleries: 7200=2hr, 14400=4hr, 28800=8hr. TIP: If it times out, just run again - gallery-dl automatically resumes from where it left off using the download archive!"
|
||||
}),
|
||||
"extra_options": ("STRING", {
|
||||
"default": "",
|
||||
"tooltip": "Additional options for gallery-dl (e.g., '--no-skip-download')"
|
||||
"multiline": True,
|
||||
"tooltip": "Advanced gallery-dl options (one per line or space-separated):\n• --limit N: Max N files to download\n• --range 1-50: Download items 1-50 only\n• --no-download: Simulate without downloading\n• --write-metadata: Save per-file JSON metadata\n• --ugoira-conv: Convert Pixiv ugoira to video\n• -v: Verbose output for debugging\n• --sleep N: Wait N seconds between requests\n• --retries N: Retry failed downloads N times\n\nFolder organization examples:\n• Kemono by post: -o directory=[\"{service}\",\"{user}\",\"{id}_{title}\"]\n• Flat by user: -o directory=[\"{category}\",\"{user}\"]\n\nSee docs: https://github.com/mikf/gallery-dl/blob/master/docs/options.md"
|
||||
}),
|
||||
"seed": ("INT", {
|
||||
"default": 0,
|
||||
"min": 0,
|
||||
"max": 0xffffffffffffffff,
|
||||
"tooltip": "Random seed to force re-execution. Click 🎲 to randomize or change manually. Useful for resuming downloads after timeout - change the seed to run again without modifying other settings."
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -738,7 +991,7 @@ class GalleryDLNode:
|
||||
FUNCTION = "execute"
|
||||
CATEGORY = "Downloaders"
|
||||
|
||||
DESCRIPTION = "Download images and media from various websites using gallery-dl"
|
||||
DESCRIPTION = "Download images and media from 100+ websites using gallery-dl. Supports Instagram, Reddit, Twitter/X, DeviantArt, Pixiv, 500px, Flickr, Pinterest, Bluesky, and more. Features: resolution filtering, duplicate detection, automatic file organization, and persistent config paths."
|
||||
|
||||
def execute(
|
||||
self,
|
||||
@@ -752,11 +1005,16 @@ class GalleryDLNode:
|
||||
use_download_archive: bool = True,
|
||||
archive_file: str = "./gallery-dl-archive.sqlite3",
|
||||
skip_videos: bool = False,
|
||||
filter_by_resolution: bool = True,
|
||||
min_image_width: int = 768,
|
||||
min_image_height: int = 768,
|
||||
extract_metadata: bool = True,
|
||||
organize_files: bool = True,
|
||||
# New advanced options
|
||||
instagram_include: str = "posts",
|
||||
download_timeout: int = 1800,
|
||||
extra_options: str = "",
|
||||
seed: int = 0,
|
||||
) -> Tuple[str, str, int, bool]:
|
||||
"""
|
||||
Execute the gallery-dl download process.
|
||||
@@ -765,6 +1023,10 @@ class GalleryDLNode:
|
||||
tuple: (output_dir, summary, download_count, success)
|
||||
"""
|
||||
try:
|
||||
# Save original config values for persistence before cleaning
|
||||
original_config_path = config_path.strip() if config_path else ""
|
||||
original_cookie_file = cookie_file.strip() if cookie_file else ""
|
||||
|
||||
# Clean up input parameters
|
||||
if url_file and url_file.strip() == "":
|
||||
url_file = None
|
||||
@@ -774,9 +1036,39 @@ class GalleryDLNode:
|
||||
cookie_file = None
|
||||
if archive_file and archive_file.strip() == "":
|
||||
archive_file = "./gallery-dl-archive.sqlite3"
|
||||
|
||||
# Save config_path and cookie_file for future use if they were provided
|
||||
if original_config_path:
|
||||
set_persistent_setting("gallery_dl", "config_path", original_config_path)
|
||||
print(f"Saved config_path for future use: {original_config_path}")
|
||||
if original_cookie_file:
|
||||
set_persistent_setting("gallery_dl", "cookie_file", original_cookie_file)
|
||||
print(f"Saved cookie_file for future use: {original_cookie_file}")
|
||||
|
||||
# Sanitize output directory
|
||||
base_output_dir = os.path.abspath(folder_paths.get_output_directory())
|
||||
|
||||
if not output_dir or output_dir.strip() == "":
|
||||
output_dir = base_output_dir
|
||||
else:
|
||||
# Handle relative paths - make relative to ComfyUI output directory
|
||||
if not os.path.isabs(output_dir):
|
||||
output_dir = os.path.join(base_output_dir, output_dir)
|
||||
|
||||
# Normalize path to resolve .. and .
|
||||
output_dir = os.path.normpath(output_dir)
|
||||
output_dir = os.path.abspath(output_dir)
|
||||
|
||||
# Security check: Ensure path is within base_output_dir
|
||||
try:
|
||||
if os.path.commonpath([base_output_dir, output_dir]) != base_output_dir:
|
||||
print(f"Warning: Path '{output_dir}' is outside the allowed output directory. Reverting to default.")
|
||||
output_dir = base_output_dir
|
||||
except ValueError:
|
||||
# Can happen on Windows if paths are on different drives
|
||||
print(f"Warning: Path '{output_dir}' is on a different drive. Reverting to default.")
|
||||
output_dir = base_output_dir
|
||||
|
||||
# Convert relative paths to absolute paths
|
||||
output_dir = os.path.abspath(output_dir)
|
||||
if archive_file:
|
||||
archive_file = os.path.abspath(archive_file)
|
||||
if config_path:
|
||||
@@ -806,9 +1098,13 @@ class GalleryDLNode:
|
||||
use_download_archive=use_download_archive,
|
||||
archive_file=archive_file,
|
||||
skip_videos=skip_videos,
|
||||
filter_by_resolution=filter_by_resolution,
|
||||
min_image_width=min_image_width,
|
||||
min_image_height=min_image_height,
|
||||
extract_metadata=extract_metadata,
|
||||
organize_files=organize_files,
|
||||
instagram_include=instagram_include,
|
||||
download_timeout=download_timeout,
|
||||
extra_options=extra_options,
|
||||
)
|
||||
|
||||
@@ -875,7 +1171,7 @@ NODE_CLASS_MAPPINGS = {
|
||||
}
|
||||
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {
|
||||
"GalleryDLDownloader": "Gallery-dl Downloader",
|
||||
"GalleryDLDownloader": "Social Media Downloader (gallery-dl)",
|
||||
}
|
||||
|
||||
# Export for ComfyUI
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+293
-21
@@ -97,6 +97,27 @@ except ImportError:
|
||||
import pathlib
|
||||
|
||||
|
||||
# Import persistent settings manager
|
||||
try:
|
||||
from ..utils.persistent_settings import get_persistent_setting, set_persistent_setting
|
||||
PERSISTENT_SETTINGS_AVAILABLE = True
|
||||
except ImportError:
|
||||
try:
|
||||
import sys
|
||||
utils_dir = os.path.dirname(os.path.dirname(__file__))
|
||||
if utils_dir not in sys.path:
|
||||
sys.path.insert(0, utils_dir)
|
||||
from utils.persistent_settings import get_persistent_setting, set_persistent_setting
|
||||
PERSISTENT_SETTINGS_AVAILABLE = True
|
||||
except ImportError:
|
||||
PERSISTENT_SETTINGS_AVAILABLE = False
|
||||
def get_persistent_setting(node_type, key, default=""):
|
||||
return default
|
||||
def set_persistent_setting(node_type, key, value):
|
||||
return False
|
||||
print("Warning: persistent_settings not available. Config paths will not persist.")
|
||||
|
||||
|
||||
# Direct Playwright support
|
||||
try:
|
||||
import playwright
|
||||
@@ -292,8 +313,15 @@ def load_site_handlers():
|
||||
elif handler_class.__name__ == "YouTubeHandler":
|
||||
return 5
|
||||
# Give site-specific handlers higher priority
|
||||
elif handler_class.__name__ in ["RedditHandler", "BskyHandler", "FlickrHandler", "ArtsyHandler"]:
|
||||
elif handler_class.__name__ in ["RedditHandler", "BskyHandler", "FlickrHandler", "ArtsyHandler", "ModelMayhemHandler", "BellazonHandler"]:
|
||||
return 10
|
||||
# Give WixHandler high priority (Wix-powered sites need special handling)
|
||||
elif handler_class.__name__ == "WixHandler":
|
||||
return 15
|
||||
# Give PortfolioHandler medium-high priority (handles generic portfolio sites)
|
||||
# Note: PortfolioHandler must be lower priority than site-specific handlers like ModelMayhemHandler
|
||||
elif handler_class.__name__ == "PortfolioHandler":
|
||||
return 50
|
||||
# Default priority
|
||||
return 100
|
||||
|
||||
@@ -326,6 +354,9 @@ class EricWebFileScraper:
|
||||
def INPUT_TYPES(cls):
|
||||
# Define common hash algorithms based on availability
|
||||
hash_algos = ["average_hash", "phash", "dhash", "whash"] if IMAGEHASH_AVAILABLE else ["none"]
|
||||
|
||||
# Load saved auth_config_path from persistent settings
|
||||
saved_auth_config = get_persistent_setting("web_scraper", "auth_config_path", "")
|
||||
|
||||
return {
|
||||
"required": {
|
||||
@@ -355,7 +386,11 @@ class EricWebFileScraper:
|
||||
# --- Crawling Options ---
|
||||
"crawl_links": ("BOOLEAN", {"default": False, "label_on": "Follow Links", "label_off": "Single Page"}),
|
||||
"crawl_depth": ("INT", {"default": 1, "min": 1, "max": 5, "step": 1, "display": "Link Crawl Depth"}),
|
||||
"max_pages": ("INT", {"default": 10, "min": 1, "max": 100, "step": 1, "display": "Max Pages to Visit"}),
|
||||
"max_pages": ("INT", {"default": 50, "min": 1, "max": 500, "step": 5, "display": "Max Pages to Visit"}),
|
||||
"crawl_subfolders": ("BOOLEAN", {"default": True, "label_on": "Subfolder Per Page", "label_off": "Single Folder"}),
|
||||
"skip_first_page_download": ("BOOLEAN", {"default": False, "label_on": "Skip Listing Page", "label_off": "Download All Pages"}),
|
||||
"link_include_pattern": ("STRING", {"default": "", "multiline": False, "placeholder": "Regex to match links (e.g. /[A-Z][a-z]+_)"}),
|
||||
"link_exclude_pattern": ("STRING", {"default": "", "multiline": False, "placeholder": "Regex: /contact|/news|/about"}),
|
||||
|
||||
# --- Media Types ---
|
||||
"download_audio": ("BOOLEAN", {"default": False, "label_on": "Extract Audio", "label_off": "Skip Audio"}),
|
||||
@@ -378,7 +413,7 @@ class EricWebFileScraper:
|
||||
"scroll_delay_ms": ("INT", {"default": 1000, "min": 50, "max": 5000, "step": 50, "display": "Scroll Delay (ms)"}),
|
||||
# --- Interactions & Auth ---
|
||||
"interaction_sequence": ("STRING", {"multiline": True, "placeholder": '[{"type": "click", "selector": "#button"}, ...]', "display": "Interaction Sequence (JSON)"}),
|
||||
"auth_config_path": ("STRING", {"default": "", "placeholder": "Path to auth_config.json"}),
|
||||
"auth_config_path": ("STRING", {"default": saved_auth_config, "placeholder": "Path to auth_config.json (saved for future use)"}),
|
||||
"save_cookies": ("BOOLEAN", {"default": False, "label_on": "Save Cookies", "label_off": "Don't Save Cookies"}),
|
||||
# --- Screenshots & Debug ---
|
||||
"take_screenshot": ("BOOLEAN", {"default": False, "label_on": "Take Screenshot", "label_off": "No Screenshot"}),
|
||||
@@ -521,6 +556,31 @@ class EricWebFileScraper:
|
||||
"""
|
||||
colored_print("--- EricWebFileScraper v0.8 ---", "94") # Bright blue
|
||||
|
||||
# Sanitize output directory
|
||||
if folder_paths:
|
||||
base_output_dir = os.path.abspath(folder_paths.get_output_directory())
|
||||
|
||||
if not output_dir or output_dir.strip() == "":
|
||||
output_dir = base_output_dir
|
||||
else:
|
||||
if not os.path.isabs(output_dir):
|
||||
output_dir = os.path.join(base_output_dir, output_dir)
|
||||
|
||||
output_dir = os.path.normpath(output_dir)
|
||||
output_dir = os.path.abspath(output_dir)
|
||||
|
||||
# Security check: Ensure path is within base_output_dir
|
||||
try:
|
||||
if os.path.commonpath([base_output_dir, output_dir]) != base_output_dir:
|
||||
print(f"Warning: Path '{output_dir}' is outside the allowed output directory. Reverting to default.")
|
||||
output_dir = base_output_dir
|
||||
except ValueError:
|
||||
print(f"Warning: Path '{output_dir}' is on a different drive. Reverting to default.")
|
||||
output_dir = base_output_dir
|
||||
else:
|
||||
# Fallback if folder_paths is not available (e.g. running standalone)
|
||||
output_dir = os.path.abspath(output_dir)
|
||||
|
||||
import asyncio
|
||||
import nest_asyncio
|
||||
|
||||
@@ -958,6 +1018,11 @@ class EricWebFileScraper:
|
||||
return ("", "", 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, json.dumps(stats))
|
||||
|
||||
self.auth_config = self.load_auth_config(auth_config_path)
|
||||
|
||||
# Save auth_config_path for future use if it was provided
|
||||
if auth_config_path and auth_config_path.strip():
|
||||
set_persistent_setting("web_scraper", "auth_config_path", auth_config_path)
|
||||
print(f"Saved auth_config_path for future use: {auth_config_path}")
|
||||
|
||||
handler_instance = self._get_handler_for_url(url)
|
||||
if handler_instance:
|
||||
@@ -1023,7 +1088,11 @@ class EricWebFileScraper:
|
||||
move_duplicates=move_duplicates,
|
||||
max_files=max_files,
|
||||
use_parallel=use_parallel,
|
||||
max_workers=max_workers
|
||||
max_workers=max_workers,
|
||||
crawl_subfolders=kwargs.get('crawl_subfolders', True),
|
||||
link_include_pattern=kwargs.get('link_include_pattern', ''),
|
||||
link_exclude_pattern=kwargs.get('link_exclude_pattern', ''),
|
||||
skip_first_page_download=kwargs.get('skip_first_page_download', False)
|
||||
)
|
||||
|
||||
# Store cache flags so _cleanup_resources() can read them later
|
||||
@@ -1450,9 +1519,20 @@ class EricWebFileScraper:
|
||||
move_duplicates=False,
|
||||
max_files=0,
|
||||
use_parallel=True,
|
||||
max_workers=4
|
||||
max_workers=4,
|
||||
crawl_subfolders=True,
|
||||
link_include_pattern="",
|
||||
link_exclude_pattern="",
|
||||
skip_first_page_download=False
|
||||
):
|
||||
"""Extract content from the current page and follow links (async version)."""
|
||||
"""Extract content from the current page and follow links (async version).
|
||||
|
||||
Args:
|
||||
crawl_subfolders: If True, save each page's images in a subfolder named after the page path
|
||||
link_include_pattern: Regex pattern - only follow links matching this pattern (if set)
|
||||
link_exclude_pattern: Regex pattern - exclude links matching this pattern
|
||||
skip_first_page_download: If True, don't download from the starting page (depth 0), only from linked pages
|
||||
"""
|
||||
if visited_urls is None:
|
||||
visited_urls = set()
|
||||
|
||||
@@ -1515,8 +1595,28 @@ class EricWebFileScraper:
|
||||
except Exception as e:
|
||||
print(f"Error during generic extraction: {e}")
|
||||
|
||||
if media_items and output_path:
|
||||
# Skip downloading from the first page if skip_first_page_download is enabled
|
||||
should_download = True
|
||||
if skip_first_page_download and current_depth == 0:
|
||||
print(f"Skipping download from listing page (depth 0): {current_url}")
|
||||
should_download = False
|
||||
|
||||
if media_items and output_path and should_download:
|
||||
print(f"Found {len(media_items)} media items, processing for download")
|
||||
|
||||
# Determine output path - use subfolder if enabled and not on base page
|
||||
page_output_path = output_path
|
||||
if crawl_subfolders and current_depth > 0:
|
||||
# Create subfolder based on page path
|
||||
parsed_current = urlparse(current_url)
|
||||
page_path = parsed_current.path.strip('/')
|
||||
if page_path:
|
||||
# Sanitize for folder name (replace / with _, remove special chars)
|
||||
subfolder_name = re.sub(r'[<>:"/\\|?*]', '_', page_path)
|
||||
subfolder_name = subfolder_name.lower().replace('/', '_')
|
||||
page_output_path = os.path.join(output_path, subfolder_name)
|
||||
os.makedirs(page_output_path, exist_ok=True)
|
||||
print(f" Saving to subfolder: {subfolder_name}")
|
||||
|
||||
download_kwargs = {
|
||||
'filename_prefix': filename_prefix,
|
||||
@@ -1538,11 +1638,11 @@ class EricWebFileScraper:
|
||||
try:
|
||||
if use_parallel:
|
||||
downloaded_data, _ = await self._process_download_queue_parallel(
|
||||
media_items, output_path, stats, max_workers=max_workers, **download_kwargs
|
||||
media_items, page_output_path, stats, max_workers=max_workers, **download_kwargs
|
||||
)
|
||||
else:
|
||||
downloaded_data, _ = await self._process_download_queue(
|
||||
media_items, output_path, stats, **download_kwargs
|
||||
media_items, page_output_path, stats, **download_kwargs
|
||||
)
|
||||
|
||||
if downloaded_data:
|
||||
@@ -1562,8 +1662,12 @@ class EricWebFileScraper:
|
||||
try:
|
||||
link_locator = page.locator("a[href]:visible")
|
||||
link_count = await link_locator.count()
|
||||
|
||||
# Use max_pages as upper limit for link extraction (was hardcoded to 50)
|
||||
max_links_to_check = max(max_pages * 2, 200) # Check more links than max_pages to allow for filtering
|
||||
print(f"Found {link_count} visible links, checking up to {min(link_count, max_links_to_check)}")
|
||||
|
||||
for i in range(min(link_count, 50)):
|
||||
for i in range(min(link_count, max_links_to_check)):
|
||||
link = link_locator.nth(i)
|
||||
href = await link.get_attribute("href")
|
||||
|
||||
@@ -1578,6 +1682,25 @@ class EricWebFileScraper:
|
||||
|
||||
if same_domain_only and link_domain != base_domain:
|
||||
continue
|
||||
|
||||
# Apply link pattern filters
|
||||
if link_include_pattern:
|
||||
try:
|
||||
if not re.search(link_include_pattern, href, re.IGNORECASE):
|
||||
continue # Skip if doesn't match include pattern
|
||||
except re.error as e:
|
||||
if debug_mode:
|
||||
print(f"Invalid include pattern regex: {e}")
|
||||
|
||||
if link_exclude_pattern:
|
||||
try:
|
||||
if re.search(link_exclude_pattern, href, re.IGNORECASE):
|
||||
if debug_mode:
|
||||
print(f" Excluding link (matched exclude pattern): {href}")
|
||||
continue # Skip if matches exclude pattern
|
||||
except re.error as e:
|
||||
if debug_mode:
|
||||
print(f"Invalid exclude pattern regex: {e}")
|
||||
|
||||
if href not in visited_urls:
|
||||
links.append(href)
|
||||
@@ -1627,7 +1750,11 @@ class EricWebFileScraper:
|
||||
move_duplicates=move_duplicates,
|
||||
max_files=max_files,
|
||||
use_parallel=use_parallel,
|
||||
max_workers=max_workers
|
||||
max_workers=max_workers,
|
||||
crawl_subfolders=crawl_subfolders,
|
||||
link_include_pattern=link_include_pattern,
|
||||
link_exclude_pattern=link_exclude_pattern,
|
||||
skip_first_page_download=skip_first_page_download
|
||||
)
|
||||
|
||||
media_items_for_download.extend(sub_items)
|
||||
@@ -2218,6 +2345,19 @@ class EricWebFileScraper:
|
||||
# Check if this is a trusted CDN URL marked by the handler
|
||||
is_trusted_cdn = item_data.get('trusted_cdn', False)
|
||||
|
||||
# Fallback: Check common CDN domains directly if not already marked
|
||||
# This handles cases where items come from cache/network monitoring
|
||||
if not is_trusted_cdn and item_domain != base_domain:
|
||||
cdn_domains = [
|
||||
'tildacdn.com', 'cloudfront.net', 'cloudflare.com',
|
||||
'akamaihd.net', 'fastly.net', 'imgix.net', 'twimg.com',
|
||||
'cdninstagram.com', 'googleapis.com', 'gstatic.com',
|
||||
'photos.modelmayhem.com', 'assets.modelmayhem.com' # ModelMayhem CDN domains
|
||||
]
|
||||
is_trusted_cdn = any(cdn in item_domain.lower() for cdn in cdn_domains)
|
||||
if is_trusted_cdn:
|
||||
item_data['trusted_cdn'] = True # Mark for future reference
|
||||
|
||||
# Debug: Show what's happening
|
||||
if item_domain != base_domain:
|
||||
print(f" Domain check: {item_domain} vs {base_domain}, trusted_cdn={is_trusted_cdn}")
|
||||
@@ -2378,6 +2518,54 @@ class EricWebFileScraper:
|
||||
print(f"Finished processing download queue. Successfully processed {len(downloaded_files_data)} new files.")
|
||||
return downloaded_files_data, downloaded_images_cache
|
||||
|
||||
def _upgrade_tilda_url(self, url):
|
||||
"""
|
||||
Upgrade Tilda CDN URLs from thumbnail/resized versions to full resolution.
|
||||
|
||||
Tilda URL patterns:
|
||||
- Thumbnail: https://optim.tildacdn.com/tild...//-/resize/720x/-/format/webp/filename.jpg.webp
|
||||
- Original: https://static.tildacdn.com/tild.../filename.jpg
|
||||
|
||||
The key is:
|
||||
1. Change optim/thb to static
|
||||
2. Remove /-/resize/.../-/format/.../ path segments
|
||||
3. Remove the .webp extension that was added
|
||||
"""
|
||||
if not url or 'tildacdn.com' not in url.lower():
|
||||
return url
|
||||
|
||||
original_url = url
|
||||
|
||||
# Replace optim/thb subdomain with static
|
||||
url = re.sub(r'https?://optim\.tildacdn\.com', 'https://static.tildacdn.com', url)
|
||||
url = re.sub(r'https?://thb\.tildacdn\.com', 'https://static.tildacdn.com', url)
|
||||
|
||||
# Remove the image transformation path segments
|
||||
# Pattern: /-/resize/NNNx/-/format/webp/ or similar
|
||||
# These need to be removed to get to the original
|
||||
url = re.sub(r'/-/resize/\d+x\d*/', '/', url)
|
||||
url = re.sub(r'/-/resize/\d+/', '/', url)
|
||||
url = re.sub(r'/-/resizeb/\d+x\d*/', '/', url)
|
||||
url = re.sub(r'/-/resizeb/\d+/', '/', url)
|
||||
url = re.sub(r'/-/format/webp/', '/', url)
|
||||
url = re.sub(r'/-/format/jpeg/', '/', url)
|
||||
url = re.sub(r'/-/quality/\d+/', '/', url)
|
||||
url = re.sub(r'/-/empty/', '/', url)
|
||||
|
||||
# Remove the added .webp extension (original is .jpg, not .jpg.webp)
|
||||
# This handles: filename.jpg.webp -> filename.jpg
|
||||
if url.endswith('.webp') and '.jpg.webp' in url:
|
||||
url = url[:-5] # Remove .webp
|
||||
elif url.endswith('.webp') and '.png.webp' in url:
|
||||
url = url[:-5] # Remove .webp
|
||||
|
||||
# Clean up any double slashes created by removal (except in https://)
|
||||
url = re.sub(r'([^:])//+', r'\1/', url)
|
||||
|
||||
if url != original_url:
|
||||
print(f" Upgraded Tilda URL: {original_url[:70]}... -> {url[:70]}...")
|
||||
|
||||
return url
|
||||
|
||||
def download_file(self, url_or_dict, output_path, index, prefix, min_width, min_height, hash_algo, download_images, download_videos):
|
||||
"""Downloads a single file, checks dimensions, calculates hash, and returns details."""
|
||||
@@ -2392,6 +2580,9 @@ class EricWebFileScraper:
|
||||
if not url:
|
||||
print("Empty URL provided")
|
||||
return None, None, None, None, None, "Missing URL"
|
||||
|
||||
# Upgrade Tilda CDN URLs to full resolution
|
||||
url = self._upgrade_tilda_url(url)
|
||||
|
||||
# Basic URL validation
|
||||
parsed_url = urlparse(url)
|
||||
@@ -2437,12 +2628,51 @@ class EricWebFileScraper:
|
||||
if is_video and not download_videos: reason = "Video download disabled"
|
||||
return None, None, None, None, None, f"Skipped ({reason}): {url}"
|
||||
|
||||
# Get item dict for additional metadata (page_url, qualities, etc.)
|
||||
item = url_or_dict if isinstance(url_or_dict, dict) else {}
|
||||
|
||||
try:
|
||||
print(f" Downloading {file_type}: {url}")
|
||||
headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36'}
|
||||
|
||||
# Add Referer header for Wix videos (required to avoid 403)
|
||||
page_url = item.get('page_url', '')
|
||||
if 'wixstatic.com' in url:
|
||||
if page_url:
|
||||
headers['Referer'] = page_url
|
||||
headers['Origin'] = '/'.join(page_url.split('/')[:3]) # Extract origin
|
||||
else:
|
||||
# Try to construct a reasonable referer from the URL
|
||||
headers['Referer'] = 'https://www.wix.com/'
|
||||
|
||||
response = requests.get(url, stream=True, timeout=30, headers=headers)
|
||||
|
||||
if response.status_code != 200:
|
||||
# Handle Wix video 403 errors with quality fallback
|
||||
if response.status_code == 403 and 'video.wixstatic.com' in url:
|
||||
qualities = item.get('qualities', {})
|
||||
original_url = item.get('original_url', '')
|
||||
|
||||
# Try fallback qualities: 720p, 480p, then original
|
||||
fallback_urls = []
|
||||
if qualities:
|
||||
for q in ['720p', '480p', '360p']:
|
||||
if q in qualities and qualities[q] != url:
|
||||
fallback_urls.append((q, qualities[q]))
|
||||
if original_url and original_url != url:
|
||||
fallback_urls.append(('original', original_url))
|
||||
|
||||
for quality, fallback_url in fallback_urls:
|
||||
print(f" Trying {quality} fallback: {fallback_url}")
|
||||
response = requests.get(fallback_url, stream=True, timeout=30, headers=headers)
|
||||
if response.status_code == 200:
|
||||
url = fallback_url # Update URL for filename generation
|
||||
print(f" Success with {quality} quality")
|
||||
break
|
||||
else:
|
||||
# All fallbacks failed
|
||||
print(f"HTTP error {response.status_code} for {url} (all quality fallbacks failed)")
|
||||
return None, None, None, None, None, f"HTTP error {response.status_code}"
|
||||
elif response.status_code != 200:
|
||||
print(f"HTTP error {response.status_code} for {url}")
|
||||
return None, None, None, None, None, f"HTTP error {response.status_code}"
|
||||
|
||||
@@ -3700,6 +3930,16 @@ class EricWebFileScraper:
|
||||
# Check if this is a trusted CDN URL marked by the handler
|
||||
is_trusted_cdn = item_data.get('trusted_cdn', False)
|
||||
|
||||
# Fallback: check common CDN domains directly (handles items from DevTools cache)
|
||||
if not is_trusted_cdn:
|
||||
cdn_domains = ['tildacdn.com', 'cloudfront.net', 'cloudflare.com', 'akamaized.net',
|
||||
'fastly.net', 'cdn.', 'assets.', 'static.', 'media.', 'images.',
|
||||
'photos.modelmayhem.com', 'assets.modelmayhem.com'] # ModelMayhem CDN
|
||||
is_trusted_cdn = any(cdn in item_domain.lower() for cdn in cdn_domains)
|
||||
if is_trusted_cdn:
|
||||
with global_lock:
|
||||
print(f" Auto-trusted CDN domain: {item_domain}")
|
||||
|
||||
# Debug: Show what's happening
|
||||
if item_domain != base_domain:
|
||||
with global_lock:
|
||||
@@ -3873,8 +4113,13 @@ class EricWebFileScraper:
|
||||
if item_url and not self.check_url_processed(item_url):
|
||||
filtered_items.append((index, item_data))
|
||||
|
||||
# If all items were already processed, return early
|
||||
if not filtered_items:
|
||||
print("All items have already been processed, skipping download.")
|
||||
return downloaded_files_data, downloaded_images_cache
|
||||
|
||||
# Process in batches for better control
|
||||
batch_size = min(max_workers * 2, len(filtered_items))
|
||||
batch_size = max(1, min(max_workers * 2, len(filtered_items))) # Ensure batch_size is at least 1
|
||||
for i in range(0, len(filtered_items), batch_size):
|
||||
# Check for cancellation before each batch
|
||||
if self.cancellation_requested or self._check_cancellation():
|
||||
@@ -4309,19 +4554,42 @@ class EricWebFileScraper:
|
||||
if not is_visible:
|
||||
continue
|
||||
|
||||
# Get image attributes
|
||||
# Get image attributes - check data-src first for lazy-loaded/full-res images
|
||||
src = await img.get_attribute("src")
|
||||
data_src = await img.get_attribute("data-src")
|
||||
data_original = await img.get_attribute("data-original")
|
||||
data_lazy_src = await img.get_attribute("data-lazy-src")
|
||||
data_full_src = await img.get_attribute("data-full-src")
|
||||
# Tilda CDN specific: data-img-zoom-url contains full-res URL for zoomable images
|
||||
data_img_zoom_url = await img.get_attribute("data-img-zoom-url")
|
||||
srcset = await img.get_attribute("srcset")
|
||||
alt = await img.get_attribute("alt") or ""
|
||||
title_attr = await img.get_attribute("title") or ""
|
||||
|
||||
# Prefer data-src attributes as they often contain full-resolution URLs
|
||||
# (e.g., Tilda CDN uses data-original or data-img-zoom-url for full-res, src for thumbnails)
|
||||
full_res_url = data_img_zoom_url or data_original or data_src or data_lazy_src or data_full_src
|
||||
|
||||
# Skip common non-content images
|
||||
if not src or any(x in src.lower() for x in ['spacer.gif', 'pixel.gif', 'transparent.gif', 'icon']):
|
||||
if not src and not full_res_url:
|
||||
continue
|
||||
if src and any(x in src.lower() for x in ['spacer.gif', 'pixel.gif', 'transparent.gif', 'icon']):
|
||||
continue
|
||||
|
||||
# Process srcset
|
||||
# Determine the best URL to use
|
||||
# Priority: data-src (full-res) > srcset highest res > src
|
||||
image_url = src
|
||||
if srcset:
|
||||
|
||||
# Check if src is a thumbnail/resized version (common CDN patterns)
|
||||
src_is_thumbnail = src and any(pattern in src for pattern in [
|
||||
'/resize/', '/-/empty/', '/thb.', '_thumb', '_small',
|
||||
'/optim.', '/thumbnail', 'w_100', 'h_100'
|
||||
])
|
||||
|
||||
# If we have a full-res data attribute, prefer that
|
||||
if full_res_url and full_res_url.startswith('http'):
|
||||
image_url = full_res_url
|
||||
elif srcset:
|
||||
high_res = self._get_highest_res_from_srcset(srcset)
|
||||
if high_res:
|
||||
image_url = high_res
|
||||
@@ -4397,14 +4665,18 @@ class EricWebFileScraper:
|
||||
print(f"Error extracting from iframe: {e}")
|
||||
|
||||
# --- Extract images from data-* attributes on any element ---
|
||||
data_attrs = ["data-large", "data-image", "data-original", "data-url"]
|
||||
# Include data-src for lazy-loaded images (Tilda, many other CMSes)
|
||||
# data-img-zoom-url is Tilda CDN's attribute for full-resolution zoomable images
|
||||
data_attrs = ["data-img-zoom-url", "data-original", "data-src", "data-large",
|
||||
"data-image", "data-url", "data-lazy-src", "data-full-src",
|
||||
"data-hi-res", "data-zoom-image"]
|
||||
for attr in data_attrs:
|
||||
elements = await page.locator(f"[{attr}]").all()
|
||||
for el in elements:
|
||||
url = await el.get_attribute(attr)
|
||||
if url and url.startswith("http"):
|
||||
attr_url = await el.get_attribute(attr)
|
||||
if attr_url and attr_url.startswith("http"):
|
||||
media_items.append({
|
||||
'url': url,
|
||||
'url': attr_url,
|
||||
'title': f"Image from {attr}",
|
||||
'source_url': url,
|
||||
'type': 'image'
|
||||
|
||||
@@ -50,6 +50,7 @@ import json
|
||||
import subprocess
|
||||
import tempfile
|
||||
import urllib.parse
|
||||
import folder_paths
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Tuple
|
||||
@@ -799,8 +800,30 @@ class YtDlpNode:
|
||||
if playlist_end and playlist_end.strip() == "":
|
||||
playlist_end = None
|
||||
|
||||
# Convert relative paths to absolute paths
|
||||
output_dir = os.path.abspath(output_dir)
|
||||
# Sanitize output directory
|
||||
base_output_dir = os.path.abspath(folder_paths.get_output_directory())
|
||||
|
||||
if not output_dir or output_dir.strip() == "":
|
||||
output_dir = base_output_dir
|
||||
else:
|
||||
# Handle relative paths - make relative to ComfyUI output directory
|
||||
if not os.path.isabs(output_dir):
|
||||
output_dir = os.path.join(base_output_dir, output_dir)
|
||||
|
||||
# Normalize path to resolve .. and .
|
||||
output_dir = os.path.normpath(output_dir)
|
||||
output_dir = os.path.abspath(output_dir)
|
||||
|
||||
# Security check: Ensure path is within base_output_dir
|
||||
try:
|
||||
if os.path.commonpath([base_output_dir, output_dir]) != base_output_dir:
|
||||
print(f"Warning: Path '{output_dir}' is outside the allowed output directory. Reverting to default.")
|
||||
output_dir = base_output_dir
|
||||
except ValueError:
|
||||
# Can happen on Windows if paths are on different drives
|
||||
print(f"Warning: Path '{output_dir}' is on a different drive. Reverting to default.")
|
||||
output_dir = base_output_dir
|
||||
|
||||
if archive_file:
|
||||
archive_file = os.path.abspath(archive_file)
|
||||
if config_path:
|
||||
@@ -908,7 +931,7 @@ NODE_CLASS_MAPPINGS = {
|
||||
}
|
||||
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {
|
||||
"YtDlpDownloader": "Yt-dlp Downloader",
|
||||
"YtDlpDownloader": "YouTube Downloader (yt-dlp)",
|
||||
}
|
||||
|
||||
# Export for ComfyUI
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
[project]
|
||||
name = "comfyui-download-tools"
|
||||
description = "ComfyUI custom nodes for large-scale media downloading via gallery-dl, yt-dlp, and advanced web scrapers. Download from 1000+ sites including Instagram, Reddit, Twitter, YouTube, TikTok, DeviantArt, Pixiv, and more."
|
||||
readme = "README.md"
|
||||
version = "0.8.5"
|
||||
license = { file = "LICENSE.md" }
|
||||
requires-python = ">=3.10"
|
||||
keywords = [
|
||||
"comfyui",
|
||||
"download",
|
||||
"gallery-dl",
|
||||
"yt-dlp",
|
||||
"web-scraper",
|
||||
"instagram",
|
||||
"youtube",
|
||||
"twitter",
|
||||
"reddit",
|
||||
"tiktok",
|
||||
"deviantart",
|
||||
"pixiv",
|
||||
"media",
|
||||
"video",
|
||||
"image"
|
||||
]
|
||||
classifiers = [
|
||||
"Development Status :: 4 - Beta",
|
||||
"Intended Audience :: Developers",
|
||||
"Intended Audience :: End Users/Desktop",
|
||||
"Operating System :: OS Independent",
|
||||
"Programming Language :: Python :: 3",
|
||||
"Programming Language :: Python :: 3.10",
|
||||
"Programming Language :: Python :: 3.11",
|
||||
"Programming Language :: Python :: 3.12",
|
||||
"Topic :: Multimedia :: Graphics",
|
||||
"Topic :: Multimedia :: Video",
|
||||
"Topic :: Internet :: WWW/HTTP"
|
||||
]
|
||||
dynamic = ["dependencies"]
|
||||
|
||||
[[project.authors]]
|
||||
name = "Eric Hiss"
|
||||
email = "eric@historic.camera"
|
||||
|
||||
[tool.setuptools.dynamic]
|
||||
dependencies = { file = ["requirements.txt"] }
|
||||
|
||||
[project.urls]
|
||||
Repository = "https://github.com/EricRollei/Download_Tools"
|
||||
Documentation = "https://github.com/EricRollei/Download_Tools/tree/main/Docs"
|
||||
"Bug Tracker" = "https://github.com/EricRollei/Download_Tools/issues"
|
||||
|
||||
[tool.comfy]
|
||||
PublisherId = "ericrollei"
|
||||
DisplayName = "Download Tools"
|
||||
Icon = ""
|
||||
includes = []
|
||||
|
||||
[tool.comfy.dependencies]
|
||||
gallery-dl = ">=1.26.0"
|
||||
yt-dlp = ">=2023.12.0"
|
||||
playwright = ">=1.40.0"
|
||||
+42
-14
@@ -1,8 +1,9 @@
|
||||
# Download Tools - Requirements
|
||||
# ComfyUI custom nodes for media downloading
|
||||
# Author: Eric Hiss (GitHub: EricRollei)
|
||||
|
||||
# ============================================================================
|
||||
# DOWNLOADER TOOLS (Required)
|
||||
# DOWNLOADER TOOLS (Required for gallery-dl and yt-dlp nodes)
|
||||
# ============================================================================
|
||||
|
||||
# Gallery-dl for downloading from various websites
|
||||
@@ -11,25 +12,44 @@ gallery-dl>=1.26.0
|
||||
# Yt-dlp for video/audio downloading
|
||||
yt-dlp>=2023.12.0
|
||||
|
||||
# Browser cookie support (for authentication)
|
||||
# Browser cookie support (for authentication with gallery-dl/yt-dlp)
|
||||
browser-cookie3>=0.19.0
|
||||
|
||||
# ============================================================================
|
||||
# WEB SCRAPER (Required for web_image_scraper node)
|
||||
# WEB SCRAPER - Core Dependencies (Required for web_image_scraper node)
|
||||
# ============================================================================
|
||||
|
||||
# Playwright for browser automation
|
||||
# Playwright for browser automation (run 'playwright install' after pip install)
|
||||
playwright>=1.40.0
|
||||
|
||||
# Scrapling for enhanced web scraping
|
||||
# Scrapling for enhanced web scraping with anti-bot detection
|
||||
scrapling>=0.2.0
|
||||
|
||||
# Image hashing for deduplication
|
||||
imagehash>=4.3.0
|
||||
|
||||
# PIL/Pillow for image processing
|
||||
Pillow>=10.0.0
|
||||
|
||||
# Async event loop nesting for ComfyUI compatibility
|
||||
nest-asyncio>=1.5.0
|
||||
|
||||
# ============================================================================
|
||||
# WEB SCRAPER - Site-Specific API Libraries (Optional but recommended)
|
||||
# ============================================================================
|
||||
|
||||
# AsyncPRAW for Reddit API access
|
||||
asyncpraw>=7.7.0
|
||||
|
||||
# AT Protocol SDK for Bluesky
|
||||
atproto>=0.0.40
|
||||
|
||||
# Instaloader for Instagram (optional, provides additional features)
|
||||
instaloader>=4.10.0
|
||||
|
||||
# aiohttp for async HTTP requests (used by some handlers)
|
||||
aiohttp>=3.9.0
|
||||
|
||||
# ============================================================================
|
||||
# CORE UTILITIES (Required)
|
||||
# ============================================================================
|
||||
@@ -46,25 +66,33 @@ colorama>=0.4.6
|
||||
jsonschema>=4.19.0
|
||||
|
||||
# ============================================================================
|
||||
# NOTES
|
||||
# INSTALLATION NOTES
|
||||
# ============================================================================
|
||||
|
||||
# 1. FFmpeg is required for yt-dlp audio extraction
|
||||
# 1. FFmpeg is required for yt-dlp audio extraction and video processing
|
||||
# Download from: https://ffmpeg.org/download.html
|
||||
# Or install via package manager:
|
||||
# - Windows: choco install ffmpeg
|
||||
# - macOS: brew install ffmpeg
|
||||
# - Linux: apt-get install ffmpeg
|
||||
|
||||
# 2. For gallery-dl authentication with private content:
|
||||
# - Use browser cookies (automatic)
|
||||
# 2. After installing playwright, run: playwright install
|
||||
# This downloads the browser binaries needed for web scraping.
|
||||
|
||||
# 3. For gallery-dl authentication with private content:
|
||||
# - Use browser cookies (automatic via browser-cookie3)
|
||||
# - Or export cookies to configs/cookies.json
|
||||
|
||||
# 3. For web_image_scraper node:
|
||||
# - After installing playwright, run: playwright install
|
||||
# - Configure authentication in configs/web_scraper/auth_config.json
|
||||
# - See configs/web_scraper/README.md for auth setup guide
|
||||
# 4. For web_image_scraper node:
|
||||
# - Configure authentication in configs/auth_config.json
|
||||
# - See CONFIG_SETUP.md for auth setup guide
|
||||
|
||||
# 4. These tools are installed system-wide and can be used from command line:
|
||||
# 5. Site-specific API setup:
|
||||
# - Reddit: Create app at https://www.reddit.com/prefs/apps
|
||||
# Add client_id, client_secret to auth_config.json
|
||||
# - Bluesky: Add username/password to auth_config.json
|
||||
# - Instagram: Add cookies or use browser session
|
||||
|
||||
# 6. These tools can also be used from command line:
|
||||
# - gallery-dl <url>
|
||||
# - yt-dlp <url>
|
||||
|
||||
@@ -315,8 +315,11 @@ class BaseSiteHandler:
|
||||
'cdninstagram.com',
|
||||
'twimg.com',
|
||||
'imgix.net',
|
||||
'tildacdn.com', # Tilda CDN (static.tildacdn.com, optim.tildacdn.com, thb.tildacdn.com)
|
||||
'cdn.',
|
||||
'static.',
|
||||
'optim.', # Optimized/resized image subdomains
|
||||
'thb.', # Thumbnail subdomains
|
||||
'assets.',
|
||||
'media.',
|
||||
'content.',
|
||||
|
||||
@@ -0,0 +1,834 @@
|
||||
"""
|
||||
Bellazon / Invision Community (IPS) Forum Handler
|
||||
|
||||
Description: Handler for Invision Power Board (IPS Community Suite) forums
|
||||
such as bellazon.com. Extracts full-resolution images from
|
||||
gallery card thumbnails that link to high-res originals.
|
||||
Author: Eric Hiss (GitHub: EricRollei)
|
||||
Contact: eric@historic.camera, eric@rollei.us
|
||||
License: Dual License (Non-Commercial and Commercial Use)
|
||||
Copyright (c) 2025 Eric Hiss. All rights reserved.
|
||||
|
||||
Dual License:
|
||||
1. Non-Commercial Use: This software is licensed under the terms of the
|
||||
Creative Commons Attribution-NonCommercial 4.0 International License.
|
||||
To view a copy of this license, visit http://creativecommons.org/licenses/by-nc/4.0/
|
||||
|
||||
2. Commercial Use: For commercial use, a separate license is required.
|
||||
Please contact Eric Hiss at eric@historic.camera or eric@rollei.us for licensing options.
|
||||
|
||||
Dependencies:
|
||||
This code depends on several third-party libraries, each with its own license.
|
||||
See CREDITS.md for a comprehensive list of dependencies and their licenses.
|
||||
|
||||
Third-party code:
|
||||
- See CREDITS.md for complete list of dependencies
|
||||
"""
|
||||
|
||||
"""
|
||||
Bellazon / IPS Community handler for the Web Image Scraper.
|
||||
|
||||
Invision Community (IPS) forums use a specific pattern for image galleries:
|
||||
- Thumbnail <img> tags have class "ipsImage_thumbnailed" and src URLs
|
||||
containing ".thumb.jpg.<hash>.jpg"
|
||||
- These thumbnails are wrapped in <a> tags whose href points to the
|
||||
full-resolution image (without ".thumb" in the path)
|
||||
- Posts are contained in article elements with class "ipsComment"
|
||||
- Forums support pagination via /page/N/ URL patterns
|
||||
|
||||
This handler:
|
||||
1. Detects IPS Community forums (bellazon.com and similar)
|
||||
2. Opens spoiler / hidden-content blocks before extraction
|
||||
3. Extracts full-resolution image URLs from the <a> wrappers around thumbnails
|
||||
and from the data-full-image attribute on <img> tags
|
||||
4. Collects YouTube / Vimeo video links found in post content
|
||||
5. Optionally crawls multiple pages of a topic
|
||||
"""
|
||||
|
||||
from site_handlers.base_handler import BaseSiteHandler
|
||||
from urllib.parse import urlparse, urljoin, unquote, urlencode, parse_qs
|
||||
import re
|
||||
import time
|
||||
import asyncio
|
||||
import traceback
|
||||
|
||||
# Safe import for Playwright types
|
||||
try:
|
||||
from playwright.async_api import Page as AsyncPage
|
||||
PLAYWRIGHT_AVAILABLE = True
|
||||
except ImportError:
|
||||
AsyncPage = None
|
||||
PLAYWRIGHT_AVAILABLE = False
|
||||
|
||||
|
||||
class BellazonHandler(BaseSiteHandler):
|
||||
"""
|
||||
Handler for Bellazon.com and other Invision Community (IPS/IPB) forums.
|
||||
|
||||
Targets the common IPS gallery-card pattern where small thumbnails
|
||||
(.thumb.jpg) link to full-resolution originals.
|
||||
"""
|
||||
|
||||
# Known IPS Community forum domains
|
||||
IPS_DOMAINS = [
|
||||
"bellazon.com",
|
||||
# Add more IPS-powered fashion/model forums here as discovered
|
||||
]
|
||||
|
||||
# Regex to detect IPS-style paginated topic URLs
|
||||
IPS_TOPIC_PATTERN = re.compile(
|
||||
r"https?://(?:www\.)?[^/]+/(?:main/)?(?:topic|forum)/\d+",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
# Regex to strip ".thumb" from an IPS upload URL to get the full-res version
|
||||
# e.g. …/filename.thumb.jpg.hash.jpg → …/filename.jpg.hash.jpg
|
||||
THUMB_STRIP_RE = re.compile(r"\.thumb\.(jpe?g|png|gif|webp)", re.IGNORECASE)
|
||||
|
||||
# Regex to match YouTube / Vimeo video URLs found in post content
|
||||
VIDEO_LINK_RE = re.compile(
|
||||
r"https?://(?:www\.)?(?:"
|
||||
r"youtube\.com/watch\?[^\s\"'<>]+"
|
||||
r"|youtu\.be/[\w-]+"
|
||||
r"|youtube\.com/embed/[\w-]+"
|
||||
r"|youtube\.com/shorts/[\w-]+"
|
||||
r"|vimeo\.com/\d+"
|
||||
r"|player\.vimeo\.com/video/\d+"
|
||||
r")",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
def __init__(self, url, scraper=None):
|
||||
super().__init__(url, scraper)
|
||||
self.debug_mode = getattr(scraper, "debug_mode", False)
|
||||
self.start_time = time.time()
|
||||
# Normalise the URL: strip /page/N/ so we have a clean base for pagination
|
||||
self.base_topic_url = self._strip_page_number(url)
|
||||
self.start_page = self._get_page_number(url)
|
||||
print(f"[BellazonHandler] Initialized for {url}")
|
||||
print(f"[BellazonHandler] Base topic URL: {self.base_topic_url}, start page: {self.start_page}")
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Detection
|
||||
# ------------------------------------------------------------------
|
||||
@classmethod
|
||||
def can_handle(cls, url):
|
||||
"""Return True for URLs on known IPS Community forums."""
|
||||
url_lower = url.lower()
|
||||
for domain in cls.IPS_DOMAINS:
|
||||
if domain in url_lower:
|
||||
return True
|
||||
# Also match generic IPS topic URL patterns with /uploads/ evidence
|
||||
if cls.IPS_TOPIC_PATTERN.search(url):
|
||||
# Only claim if the domain looks like a forum
|
||||
parsed = urlparse(url)
|
||||
path = parsed.path.lower()
|
||||
if "/topic/" in path or "/forum/" in path:
|
||||
# Heuristic: could be IPS, but only claim known domains
|
||||
# to avoid false positives with other forum software
|
||||
pass
|
||||
return False
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Trusted domains
|
||||
# ------------------------------------------------------------------
|
||||
def get_trusted_domains(self) -> list:
|
||||
"""IPS forums host uploads on the same domain."""
|
||||
return [self.domain, f"www.{self.domain}"]
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Content directory
|
||||
# ------------------------------------------------------------------
|
||||
def get_content_directory(self):
|
||||
"""Generate a meaningful output directory based on the topic."""
|
||||
parsed = urlparse(self.url)
|
||||
path_parts = [p for p in parsed.path.strip("/").split("/") if p]
|
||||
|
||||
base_dir = self._sanitize_directory_name(self.domain.split(".")[0])
|
||||
|
||||
# Try to extract topic slug (e.g. "88521-clémence-navarro")
|
||||
topic_dir = "general"
|
||||
for i, part in enumerate(path_parts):
|
||||
if part == "topic" and i + 1 < len(path_parts):
|
||||
topic_dir = self._sanitize_directory_name(
|
||||
unquote(path_parts[i + 1])
|
||||
)
|
||||
break
|
||||
|
||||
return (base_dir, topic_dir)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Main Playwright extraction (async)
|
||||
# ------------------------------------------------------------------
|
||||
async def extract_with_direct_playwright(self, page, **kwargs) -> list:
|
||||
"""
|
||||
Extract full-resolution images from ALL pages of an IPS forum topic.
|
||||
|
||||
Workflow:
|
||||
1. Detect total page count from the pagination controls
|
||||
2. Extract images from the current (or first) page
|
||||
3. Navigate to each subsequent page and extract images
|
||||
4. Respect the scraper's max_pages setting to cap how far we go
|
||||
"""
|
||||
print(f"[BellazonHandler] Starting multi-page extraction …")
|
||||
|
||||
# Determine how many pages to visit
|
||||
# The scraper passes max_pages via kwargs from the UI "Max Pages to Visit" control
|
||||
max_pages = kwargs.get("max_pages", 500)
|
||||
# Also check the scraper instance for max_pages if not in kwargs
|
||||
if max_pages <= 1 and hasattr(self, "scraper") and self.scraper:
|
||||
max_pages = getattr(self.scraper, "max_pages", 500)
|
||||
|
||||
# Detect total number of pages from IPS pagination
|
||||
total_pages = await self._detect_total_pages(page)
|
||||
print(f"[BellazonHandler] Detected {total_pages} total page(s) in topic")
|
||||
|
||||
# Determine page range
|
||||
pages_to_visit = min(total_pages, max_pages)
|
||||
start = self.start_page
|
||||
end = min(start + pages_to_visit - 1, total_pages)
|
||||
print(f"[BellazonHandler] Will scrape pages {start} through {end} "
|
||||
f"(max_pages={max_pages})")
|
||||
|
||||
all_media_items = []
|
||||
seen_urls = set()
|
||||
|
||||
for page_num in range(start, end + 1):
|
||||
# Navigate to the correct page (skip navigation for the first page
|
||||
# since we're already on it)
|
||||
if page_num != self.start_page:
|
||||
page_url = self._build_page_url(page_num)
|
||||
print(f"[BellazonHandler] Navigating to page {page_num}/{end}: {page_url}")
|
||||
try:
|
||||
await page.goto(page_url, timeout=30000, wait_until="load")
|
||||
# Small delay to let IPS JS / lazy-loading settle
|
||||
await page.wait_for_timeout(1500)
|
||||
except Exception as e:
|
||||
print(f"[BellazonHandler] Failed to navigate to page {page_num}: {e}")
|
||||
continue
|
||||
|
||||
# Extract images from this page
|
||||
page_items = await self._extract_images_from_current_page(
|
||||
page, page_num, seen_urls
|
||||
)
|
||||
all_media_items.extend(page_items)
|
||||
print(f"[BellazonHandler] Page {page_num}: {len(page_items)} images "
|
||||
f"(running total: {len(all_media_items)})")
|
||||
|
||||
# Safety: don't hammer the server
|
||||
if page_num < end:
|
||||
await page.wait_for_timeout(500)
|
||||
|
||||
# --- Fallback: use base handler if we found nothing ---
|
||||
if not all_media_items:
|
||||
print("[BellazonHandler] No items from IPS-specific extraction, "
|
||||
"falling back to base handler")
|
||||
all_media_items = await super().extract_with_direct_playwright(
|
||||
page, **kwargs
|
||||
)
|
||||
|
||||
print(f"[BellazonHandler] Total images extracted across all pages: "
|
||||
f"{len(all_media_items)}")
|
||||
|
||||
# Final safety pass: strip any remaining .thumb. URLs and deduplicate
|
||||
all_media_items = await self.post_process(all_media_items)
|
||||
print(f"[BellazonHandler] After post-processing: {len(all_media_items)} images")
|
||||
|
||||
return all_media_items
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Single-page image extraction (called per page)
|
||||
# ------------------------------------------------------------------
|
||||
async def _extract_images_from_current_page(
|
||||
self, page, page_num: int, seen_urls: set
|
||||
) -> list:
|
||||
"""Extract full-res images from the currently-loaded IPS page.
|
||||
|
||||
IPS Community HTML structure (confirmed on bellazon.com):
|
||||
─────────────────────────────────────────────────────────
|
||||
Every user-uploaded image appears as:
|
||||
|
||||
<a class="ipsAttachLink ipsAttachLink_image"
|
||||
href="https://…/uploads/…/name.jpg.HASH_FULL.jpg"
|
||||
data-fileid="12345" data-fileext="jpg">
|
||||
<img class="ipsImage ipsImage_thumbnailed"
|
||||
data-fileid="12345"
|
||||
src="https://…/uploads/…/name.thumb.jpg.HASH_THUMB.jpg"
|
||||
width="241" alt="…">
|
||||
</a>
|
||||
|
||||
KEY INSIGHT: the hash in the thumb URL is DIFFERENT from the hash
|
||||
in the full-res URL. You cannot derive full-res from the thumb
|
||||
src via regex – you MUST read the parent <a> href.
|
||||
|
||||
Strategy:
|
||||
1. Grab every <a class*="ipsAttachLink_image"> href (full-res)
|
||||
2. For any remaining content <img> not already covered, accept
|
||||
it ONLY if its src does NOT contain ".thumb." (direct-linked
|
||||
full-res images that some users paste).
|
||||
3. NEVER add a URL that contains ".thumb." – it always points
|
||||
to a low-res thumbnail with a wrong hash.
|
||||
"""
|
||||
media_items = []
|
||||
|
||||
try:
|
||||
# Wait for post content to be present
|
||||
try:
|
||||
await page.wait_for_selector(
|
||||
"article.ipsComment, div.ipsComment, div.cPost",
|
||||
timeout=15000,
|
||||
)
|
||||
except Exception:
|
||||
print(f"[BellazonHandler] Page {page_num}: could not find IPS "
|
||||
"post containers, proceeding with full-page extraction")
|
||||
|
||||
# --- Reveal spoiler / hidden content blocks ---
|
||||
await self._open_spoilers(page, page_num)
|
||||
|
||||
# --- JavaScript extraction ---
|
||||
extracted_items = await page.evaluate("""
|
||||
() => {
|
||||
const results = [];
|
||||
const seen = new Set(); // full-res URLs already added
|
||||
const seenThumbs = new Set(); // thumb srcs we've resolved via <a>
|
||||
|
||||
// Helper: is this a content image URL (not UI junk)?
|
||||
const isContentUrl = (url) => {
|
||||
if (!url) return false;
|
||||
const lower = url.toLowerCase();
|
||||
// Must be an image
|
||||
if (!/\\.(jpe?g|png|gif|webp)/i.test(lower)) return false;
|
||||
// Reject common UI images
|
||||
if (/\\/emoticons\\/|default_photo|\\/avatars?\\/|\\/core_|\\/emoji\\/|favicon|logo/i.test(lower)) return false;
|
||||
return true;
|
||||
};
|
||||
|
||||
// Helper: add to results if the full URL is not yet seen
|
||||
const addIfNew = (fullUrl, thumbSrc, img) => {
|
||||
if (!fullUrl || seen.has(fullUrl)) return;
|
||||
if (!isContentUrl(fullUrl)) return;
|
||||
// REJECT any URL that still contains .thumb.
|
||||
if (fullUrl.includes('.thumb.')) return;
|
||||
seen.add(fullUrl);
|
||||
if (thumbSrc) seenThumbs.add(thumbSrc);
|
||||
results.push({
|
||||
url: fullUrl,
|
||||
thumb_url: thumbSrc || '',
|
||||
alt: img ? (img.alt || '') : '',
|
||||
width: img ? (img.naturalWidth || 0) : 0,
|
||||
height: img ? (img.naturalHeight || 0) : 0,
|
||||
data_fileid: img ? (img.getAttribute('data-fileid') || '') : '',
|
||||
});
|
||||
};
|
||||
|
||||
// ── Strategy 1 (PRIMARY): <a> links around thumbnails ──
|
||||
// The <a href> IS the authoritative full-res URL.
|
||||
// Also handles lightbox links (data-ipslightbox).
|
||||
document.querySelectorAll(
|
||||
'a.ipsAttachLink_image[href], ' +
|
||||
'a.ipsAttachLink[href], ' +
|
||||
'a[data-ipslightbox][href]'
|
||||
).forEach(link => {
|
||||
const href = link.href;
|
||||
if (!href) return;
|
||||
const img = link.querySelector('img');
|
||||
const thumbSrc = img ? img.src : '';
|
||||
if (thumbSrc) seenThumbs.add(thumbSrc);
|
||||
addIfNew(href, thumbSrc, img);
|
||||
});
|
||||
|
||||
// Also catch thumbnails with data-fileid whose parent
|
||||
// <a> might not have the ipsAttachLink class
|
||||
document.querySelectorAll(
|
||||
'a[href] img.ipsImage_thumbnailed, ' +
|
||||
'a[href] img[data-fileid]'
|
||||
).forEach(img => {
|
||||
const link = img.closest('a');
|
||||
if (!link) return;
|
||||
const href = link.href;
|
||||
if (!href) return;
|
||||
const thumbSrc = img.src || '';
|
||||
if (thumbSrc) seenThumbs.add(thumbSrc);
|
||||
addIfNew(href, thumbSrc, img);
|
||||
});
|
||||
|
||||
// ── Strategy 2: data-full-image attribute ──
|
||||
// IPS sometimes puts the full-res URL in a data
|
||||
// attribute on the <img> itself.
|
||||
document.querySelectorAll('img[data-full-image]').forEach(img => {
|
||||
const fullUrl = img.getAttribute('data-full-image');
|
||||
const thumbSrc = img.src || '';
|
||||
if (thumbSrc) seenThumbs.add(thumbSrc);
|
||||
addIfNew(fullUrl, thumbSrc, img);
|
||||
});
|
||||
|
||||
// ── Strategy 3: Non-thumbnail content images ──
|
||||
// Some users paste direct full-res URLs into posts.
|
||||
// These imgs will NOT have .thumb. in their src and
|
||||
// will NOT have been caught by earlier strategies.
|
||||
document.querySelectorAll(
|
||||
'div[data-role="commentContent"] img, ' +
|
||||
'div.ipsType_richText img, ' +
|
||||
'div.cPost_contentWrap img'
|
||||
).forEach(img => {
|
||||
const src = img.src;
|
||||
if (!src || src.startsWith('data:')) return;
|
||||
// Skip if we already resolved this thumb via <a> href
|
||||
if (seenThumbs.has(src)) return;
|
||||
if (seen.has(src)) return;
|
||||
// REJECT any remaining .thumb. URL – we have no way
|
||||
// to derive the correct full-res hash from it
|
||||
if (src.includes('.thumb.')) return;
|
||||
// Skip tiny UI images
|
||||
if (img.naturalWidth && img.naturalWidth < 80) return;
|
||||
if (img.naturalHeight && img.naturalHeight < 80) return;
|
||||
// Skip avatars and profile photos
|
||||
if (img.closest('.ipsUserPhoto, .ipsPhotoPanel, .cAuthorPane')) return;
|
||||
// Skip quoted content to avoid duplicates
|
||||
if (img.closest('blockquote, .ipsQuote')) return;
|
||||
addIfNew(src, '', img);
|
||||
});
|
||||
|
||||
return results;
|
||||
}
|
||||
""")
|
||||
|
||||
if extracted_items:
|
||||
print(f"[BellazonHandler] Page {page_num}: JS extracted "
|
||||
f"{len(extracted_items)} full-res image URLs")
|
||||
|
||||
for item in extracted_items:
|
||||
url = item.get("url", "")
|
||||
if not url:
|
||||
continue
|
||||
|
||||
# ABSOLUTE SAFETY: reject any URL still containing .thumb.
|
||||
# (The JS already filters these, but belt-and-suspenders)
|
||||
if ".thumb." in url.lower():
|
||||
if self.debug_mode:
|
||||
print(f"[BellazonHandler] REJECTED thumb URL: {url[:80]}…")
|
||||
continue
|
||||
|
||||
if url in seen_urls:
|
||||
continue
|
||||
seen_urls.add(url)
|
||||
|
||||
# Ensure absolute URL
|
||||
if not url.startswith("http"):
|
||||
url = urljoin(self.url, url)
|
||||
|
||||
# Determine title from alt text or filename
|
||||
alt = item.get("alt", "")
|
||||
title = self._clean_title(alt) if alt else self._title_from_url(url)
|
||||
|
||||
media_items.append({
|
||||
"url": url,
|
||||
"type": "image",
|
||||
"title": title,
|
||||
"alt": alt,
|
||||
"width": item.get("width", 0),
|
||||
"height": item.get("height", 0),
|
||||
"source_url": self.url,
|
||||
"trusted_cdn": True,
|
||||
"data_fileid": item.get("data_fileid", ""),
|
||||
"thumb_url": item.get("thumb_url", ""),
|
||||
})
|
||||
|
||||
# --- Collect video links (YouTube / Vimeo) ---
|
||||
video_items = await self._extract_video_links(page, page_num, seen_urls)
|
||||
if video_items:
|
||||
media_items.extend(video_items)
|
||||
|
||||
except Exception as e:
|
||||
print(f"[BellazonHandler] Error during extraction: {e}")
|
||||
traceback.print_exc()
|
||||
|
||||
return media_items
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Spoiler / hidden-content handling
|
||||
# ------------------------------------------------------------------
|
||||
async def _open_spoilers(self, page, page_num: int) -> int:
|
||||
"""
|
||||
Open all spoiler / hidden-content blocks on the current page so
|
||||
that their images become visible in the DOM.
|
||||
|
||||
IPS Community uses HTML5 <details> elements for spoilers:
|
||||
<details class="ipsRichTextBox">
|
||||
<summary class="ipsRichTextBox__title">
|
||||
<p>Spoiler Nudity</p> ← or "Spoiler", "Reveal hidden contents", etc.
|
||||
</summary>
|
||||
… hidden images / content …
|
||||
</details>
|
||||
|
||||
Approach:
|
||||
1. Use JavaScript to programmatically set the `open` attribute on
|
||||
every <details> element – this is instantaneous and avoids
|
||||
click-timing issues.
|
||||
2. Fall back to clicking <summary> elements if JS approach fails.
|
||||
3. Wait briefly for any lazy-loaded images inside spoilers to load.
|
||||
|
||||
Returns the number of spoiler blocks that were opened.
|
||||
"""
|
||||
try:
|
||||
opened = await page.evaluate("""
|
||||
() => {
|
||||
let count = 0;
|
||||
// Open ALL <details> elements (IPS spoiler blocks)
|
||||
document.querySelectorAll('details').forEach(d => {
|
||||
if (!d.open) {
|
||||
d.open = true;
|
||||
count++;
|
||||
}
|
||||
});
|
||||
// Also look for IPS-specific spoiler toggles that might
|
||||
// not use <details> (older IPS versions)
|
||||
document.querySelectorAll(
|
||||
'.ipsSpoiler_header, ' +
|
||||
'[data-action="toggleSpoiler"], ' +
|
||||
'.ipsStyle_spoiler'
|
||||
).forEach(btn => {
|
||||
const container = btn.closest('.ipsSpoiler, [data-ipsSpoiler]');
|
||||
if (container) {
|
||||
container.classList.add('ipsSpoiler_open');
|
||||
container.style.display = '';
|
||||
// Un-hide the content inside
|
||||
const content = container.querySelector(
|
||||
'.ipsSpoiler_contents, .ipsSpoiler_content'
|
||||
);
|
||||
if (content) {
|
||||
content.style.display = '';
|
||||
content.style.visibility = 'visible';
|
||||
content.style.maxHeight = 'none';
|
||||
}
|
||||
count++;
|
||||
}
|
||||
});
|
||||
return count;
|
||||
}
|
||||
""")
|
||||
|
||||
if opened > 0:
|
||||
print(f"[BellazonHandler] Page {page_num}: opened {opened} "
|
||||
f"spoiler/hidden-content block(s)")
|
||||
# Wait for any lazy-loaded images inside spoilers to start loading
|
||||
await page.wait_for_timeout(1000)
|
||||
|
||||
# Trigger lazy loading by scrolling spoiler content into view
|
||||
await page.evaluate("""
|
||||
() => {
|
||||
document.querySelectorAll(
|
||||
'details[open] img[loading="lazy"], ' +
|
||||
'.ipsSpoiler_open img[loading="lazy"]'
|
||||
).forEach(img => {
|
||||
img.scrollIntoView({ behavior: 'instant', block: 'center' });
|
||||
});
|
||||
}
|
||||
""")
|
||||
await page.wait_for_timeout(500)
|
||||
|
||||
return opened
|
||||
|
||||
except Exception as e:
|
||||
print(f"[BellazonHandler] Page {page_num}: error opening spoilers: {e}")
|
||||
# Fallback: try clicking <summary> elements directly
|
||||
try:
|
||||
summaries = page.locator("details:not([open]) > summary")
|
||||
count = await summaries.count()
|
||||
if count > 0:
|
||||
print(f"[BellazonHandler] Page {page_num}: clicking "
|
||||
f"{count} <summary> element(s) as fallback")
|
||||
for i in range(count):
|
||||
try:
|
||||
await summaries.nth(i).click(timeout=2000)
|
||||
except Exception:
|
||||
pass
|
||||
await page.wait_for_timeout(800)
|
||||
return count
|
||||
except Exception:
|
||||
pass
|
||||
return 0
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Video link collection
|
||||
# ------------------------------------------------------------------
|
||||
async def _extract_video_links(
|
||||
self, page, page_num: int, seen_urls: set
|
||||
) -> list:
|
||||
"""
|
||||
Collect YouTube and Vimeo video URLs found in post content.
|
||||
|
||||
These are returned as media items with type="video" so the
|
||||
scraper can log them. They won't be downloaded by the image
|
||||
scraper but will appear in the results / metadata for the user
|
||||
to process with the yt-dlp node if desired.
|
||||
"""
|
||||
video_items = []
|
||||
try:
|
||||
raw_links = await page.evaluate("""
|
||||
() => {
|
||||
const links = new Set();
|
||||
// 1. <a> tags linking to YouTube / Vimeo
|
||||
document.querySelectorAll(
|
||||
'div[data-role="commentContent"] a[href], ' +
|
||||
'div.ipsType_richText a[href], ' +
|
||||
'div.cPost_contentWrap a[href]'
|
||||
).forEach(a => {
|
||||
const href = a.href || '';
|
||||
if (/youtu\.?be|youtube\.com|vimeo\.com/i.test(href)) {
|
||||
links.add(href);
|
||||
}
|
||||
});
|
||||
// 2. Embedded iframes (YouTube / Vimeo embeds)
|
||||
document.querySelectorAll(
|
||||
'iframe[src]'
|
||||
).forEach(iframe => {
|
||||
const src = iframe.src || '';
|
||||
if (/youtube\.com\/embed|player\.vimeo\.com/i.test(src)) {
|
||||
links.add(src);
|
||||
}
|
||||
});
|
||||
// 3. IPS oembed containers
|
||||
document.querySelectorAll(
|
||||
'[data-embed-src]'
|
||||
).forEach(el => {
|
||||
const src = el.getAttribute('data-embed-src') || '';
|
||||
if (/youtu\.?be|youtube\.com|vimeo\.com/i.test(src)) {
|
||||
links.add(src);
|
||||
}
|
||||
});
|
||||
return Array.from(links);
|
||||
}
|
||||
""")
|
||||
|
||||
for link in raw_links:
|
||||
if not link or link in seen_urls:
|
||||
continue
|
||||
seen_urls.add(link)
|
||||
|
||||
# Normalise embed URLs to standard watch URLs
|
||||
clean_url = self._normalise_video_url(link)
|
||||
|
||||
video_items.append({
|
||||
"url": clean_url,
|
||||
"type": "video",
|
||||
"title": f"Video: {clean_url}",
|
||||
"alt": "",
|
||||
"width": 0,
|
||||
"height": 0,
|
||||
"source_url": self.url,
|
||||
"trusted_cdn": True,
|
||||
"platform": "youtube" if "youtu" in clean_url.lower() else "vimeo",
|
||||
})
|
||||
|
||||
if video_items:
|
||||
print(f"[BellazonHandler] Page {page_num}: collected "
|
||||
f"{len(video_items)} video link(s)")
|
||||
|
||||
except Exception as e:
|
||||
if self.debug_mode:
|
||||
print(f"[BellazonHandler] Page {page_num}: error collecting "
|
||||
f"video links: {e}")
|
||||
|
||||
return video_items
|
||||
|
||||
@staticmethod
|
||||
def _normalise_video_url(url: str) -> str:
|
||||
"""
|
||||
Convert embed / shortened video URLs to canonical watch URLs.
|
||||
|
||||
Examples:
|
||||
https://www.youtube.com/embed/ABC123 → https://www.youtube.com/watch?v=ABC123
|
||||
https://youtu.be/ABC123 → https://www.youtube.com/watch?v=ABC123
|
||||
https://player.vimeo.com/video/12345 → https://vimeo.com/12345
|
||||
"""
|
||||
# YouTube embed → watch
|
||||
m = re.search(r"youtube\.com/embed/([\w-]+)", url)
|
||||
if m:
|
||||
return f"https://www.youtube.com/watch?v={m.group(1)}"
|
||||
# YouTube short URL
|
||||
m = re.search(r"youtu\.be/([\w-]+)", url)
|
||||
if m:
|
||||
return f"https://www.youtube.com/watch?v={m.group(1)}"
|
||||
# YouTube shorts
|
||||
m = re.search(r"youtube\.com/shorts/([\w-]+)", url)
|
||||
if m:
|
||||
return f"https://www.youtube.com/watch?v={m.group(1)}"
|
||||
# Vimeo player embed
|
||||
m = re.search(r"player\.vimeo\.com/video/(\d+)", url)
|
||||
if m:
|
||||
return f"https://vimeo.com/{m.group(1)}"
|
||||
return url
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Pagination helpers
|
||||
# ------------------------------------------------------------------
|
||||
async def _detect_total_pages(self, page) -> int:
|
||||
"""
|
||||
Read the IPS pagination controls to determine the total number of
|
||||
pages in the current topic.
|
||||
|
||||
IPS pagination HTML typically looks like:
|
||||
<li class="ipsPagination_last">
|
||||
<a href="…/page/16/" …>Last</a>
|
||||
</li>
|
||||
or the pagination bar contains numbered links like:
|
||||
<a …>16</a>
|
||||
We also look for the "PAGE X OF Y" text.
|
||||
"""
|
||||
try:
|
||||
total = await page.evaluate("""
|
||||
() => {
|
||||
// Strategy 1: "PAGE X OF Y" text (e.g. "PAGE 2 OF 16")
|
||||
const pageOfText = document.body.innerText.match(
|
||||
/PAGE\\s+(\\d+)\\s+OF\\s+(\\d+)/i
|
||||
);
|
||||
if (pageOfText) return parseInt(pageOfText[2], 10);
|
||||
|
||||
// Strategy 2: Last-page link
|
||||
const lastLink = document.querySelector(
|
||||
'li.ipsPagination_last a[href], ' +
|
||||
'a.ipsPagination_last[href]'
|
||||
);
|
||||
if (lastLink) {
|
||||
const href = lastLink.href;
|
||||
const m = href.match(/\\/page\\/(\\d+)/);
|
||||
if (m) return parseInt(m[1], 10);
|
||||
}
|
||||
|
||||
// Strategy 3: Highest numbered page link in the paginator
|
||||
let maxPage = 1;
|
||||
document.querySelectorAll(
|
||||
'ul.ipsPagination li a, ' +
|
||||
'div.ipsPagination a'
|
||||
).forEach(a => {
|
||||
const href = a.href || '';
|
||||
const m = href.match(/\\/page\\/(\\d+)/);
|
||||
if (m) {
|
||||
const n = parseInt(m[1], 10);
|
||||
if (n > maxPage) maxPage = n;
|
||||
}
|
||||
// Also check link text for plain numbers
|
||||
const txt = a.textContent.trim();
|
||||
if (/^\\d+$/.test(txt)) {
|
||||
const n = parseInt(txt, 10);
|
||||
if (n > maxPage) maxPage = n;
|
||||
}
|
||||
});
|
||||
return maxPage;
|
||||
}
|
||||
""")
|
||||
return max(1, total)
|
||||
except Exception as e:
|
||||
print(f"[BellazonHandler] Could not detect page count: {e}")
|
||||
return 1
|
||||
|
||||
def _strip_page_number(self, url: str) -> str:
|
||||
"""
|
||||
Remove /page/N/ from an IPS topic URL to get the base URL.
|
||||
e.g. …/topic/88521-name/page/2/ → …/topic/88521-name/
|
||||
"""
|
||||
return re.sub(r"/page/\d+/?", "/", url).rstrip("/") + "/"
|
||||
|
||||
def _get_page_number(self, url: str) -> int:
|
||||
"""Extract the page number from a URL, defaulting to 1."""
|
||||
m = re.search(r"/page/(\d+)", url)
|
||||
return int(m.group(1)) if m else 1
|
||||
|
||||
def _build_page_url(self, page_num: int) -> str:
|
||||
"""
|
||||
Build the URL for a specific page number of the topic.
|
||||
Page 1 uses the base URL (no /page/ suffix).
|
||||
"""
|
||||
base = self.base_topic_url.rstrip("/")
|
||||
if page_num <= 1:
|
||||
return base + "/"
|
||||
return f"{base}/page/{page_num}/"
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Post-processing
|
||||
# ------------------------------------------------------------------
|
||||
async def post_process(self, media_items):
|
||||
"""
|
||||
Final safety pass before items are sent to the download queue.
|
||||
|
||||
Rules:
|
||||
1. REJECT any URL that still contains ".thumb." – these are
|
||||
low-res thumbnails and their hash is different from the
|
||||
full-res version, so stripping .thumb. would produce a 404.
|
||||
2. Deduplicate by URL.
|
||||
3. Remove common non-content images (emoticons, avatars, etc.).
|
||||
"""
|
||||
upgraded = []
|
||||
seen = set()
|
||||
rejected_thumbs = 0
|
||||
|
||||
for item in media_items:
|
||||
url = item.get("url", "")
|
||||
if not url:
|
||||
continue
|
||||
|
||||
# HARD REJECT: any URL containing .thumb. is a thumbnail
|
||||
# with a wrong hash – do NOT try to "fix" it, just drop it
|
||||
if ".thumb." in url.lower():
|
||||
rejected_thumbs += 1
|
||||
continue
|
||||
|
||||
# Deduplicate
|
||||
if url in seen:
|
||||
continue
|
||||
seen.add(url)
|
||||
|
||||
# Filter out common non-content images
|
||||
url_lower = url.lower()
|
||||
if any(p in url_lower for p in [
|
||||
"/emoticons/", "/emoji/", "default_photo",
|
||||
"profile_photo", "/avatars/", "/reputation/",
|
||||
"/core_", "favicon",
|
||||
]):
|
||||
continue
|
||||
|
||||
upgraded.append(item)
|
||||
|
||||
if rejected_thumbs:
|
||||
print(f"[BellazonHandler] post_process rejected {rejected_thumbs} "
|
||||
f"remaining .thumb. URLs")
|
||||
|
||||
return upgraded
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ------------------------------------------------------------------
|
||||
def _clean_title(self, alt_text: str) -> str:
|
||||
"""
|
||||
Clean an IPS image alt text to produce a readable title.
|
||||
IPS alt text often looks like:
|
||||
'filename.thumb.jpg.hash.jpg'
|
||||
"""
|
||||
if not alt_text:
|
||||
return "Untitled"
|
||||
# Remove hash suffixes (e.g. .bbef56b4...695b.jpg)
|
||||
cleaned = re.sub(r"\.[a-f0-9]{20,}\.(jpe?g|png|gif|webp)$", "", alt_text, flags=re.IGNORECASE)
|
||||
# Remove .thumb
|
||||
cleaned = re.sub(r"\.thumb", "", cleaned, flags=re.IGNORECASE)
|
||||
# Remove file extension
|
||||
cleaned = re.sub(r"\.(jpe?g|png|gif|webp)$", "", cleaned, flags=re.IGNORECASE)
|
||||
# Replace underscores/dashes with spaces
|
||||
cleaned = cleaned.replace("_", " ").replace("-", " ")
|
||||
# Collapse whitespace
|
||||
cleaned = re.sub(r"\s+", " ", cleaned).strip()
|
||||
return cleaned if cleaned else "Untitled"
|
||||
|
||||
def _title_from_url(self, url: str) -> str:
|
||||
"""Extract a human-readable title from a URL path."""
|
||||
try:
|
||||
path = urlparse(url).path
|
||||
filename = path.rsplit("/", 1)[-1] if "/" in path else path
|
||||
return self._clean_title(unquote(filename))
|
||||
except Exception:
|
||||
return "Untitled"
|
||||
|
||||
@staticmethod
|
||||
def _strip_thumb(url: str) -> str:
|
||||
"""Remove .thumb from an IPS upload URL to get the full-res version."""
|
||||
return BellazonHandler.THUMB_STRIP_RE.sub(r".\1", url)
|
||||
@@ -855,6 +855,12 @@ class GenericWebsiteWithAuthHandler(BaseSiteHandler):
|
||||
|
||||
# Update the URL in the item
|
||||
item['url'] = url
|
||||
|
||||
# Mark CDN domains as trusted to allow cross-domain downloads
|
||||
# This is critical for sites like Tilda that serve images from tildacdn.com
|
||||
if self.is_trusted_domain(url):
|
||||
item['trusted_cdn'] = True
|
||||
|
||||
unique_items.append(item)
|
||||
seen_urls.add(clean_url)
|
||||
|
||||
@@ -981,18 +987,52 @@ class GenericWebsiteWithAuthHandler(BaseSiteHandler):
|
||||
// Get all image elements
|
||||
const imgElements = document.querySelectorAll('img');
|
||||
imgElements.forEach(img => {
|
||||
// Check src attribute
|
||||
if (img.src && img.src.startsWith('http')) {
|
||||
// Check for data-src first (lazy-loaded images like Tilda CDN)
|
||||
// These often contain the full-resolution original
|
||||
// Tilda uses data-img-zoom-url and data-original for full-res zoomable images
|
||||
const dataSrc = img.getAttribute('data-img-zoom-url') ||
|
||||
img.getAttribute('data-original') ||
|
||||
img.getAttribute('data-src') ||
|
||||
img.getAttribute('data-lazy-src') ||
|
||||
img.getAttribute('data-full-src') ||
|
||||
img.getAttribute('data-image');
|
||||
|
||||
if (dataSrc && dataSrc.startsWith('http')) {
|
||||
items.push({
|
||||
url: img.src,
|
||||
url: dataSrc,
|
||||
alt: img.alt || '',
|
||||
title: img.title || '',
|
||||
width: img.naturalWidth || img.width || 0,
|
||||
height: img.naturalHeight || img.height || 0,
|
||||
type: 'image'
|
||||
type: 'image',
|
||||
isFullRes: true // Mark as likely full-res from data attribute
|
||||
});
|
||||
}
|
||||
|
||||
// Check src attribute
|
||||
if (img.src && img.src.startsWith('http')) {
|
||||
// Skip if it's a placeholder/thumbnail URL from known CDNs with resize markers
|
||||
const isResized = img.src.includes('/resize/') ||
|
||||
img.src.includes('/-/empty/') ||
|
||||
img.src.includes('/thb.') ||
|
||||
img.src.includes('_thumb') ||
|
||||
img.src.includes('_small');
|
||||
|
||||
// If we already have a data-src for this image, prefer that
|
||||
// Only add src if it looks like a full-res version or no data-src exists
|
||||
if (!dataSrc || !isResized) {
|
||||
items.push({
|
||||
url: img.src,
|
||||
alt: img.alt || '',
|
||||
title: img.title || '',
|
||||
width: img.naturalWidth || img.width || 0,
|
||||
height: img.naturalHeight || img.height || 0,
|
||||
type: 'image',
|
||||
isFullRes: !isResized
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Check srcset attribute
|
||||
if (img.srcset) {
|
||||
const srcsetParts = img.srcset.split(',');
|
||||
@@ -1089,10 +1129,12 @@ class GenericWebsiteWithAuthHandler(BaseSiteHandler):
|
||||
alt_text = item.get('alt', '') or page_title
|
||||
title = item.get('title', '') or alt_text or page_title
|
||||
|
||||
# Filter by size if available
|
||||
# Filter by size if available - but skip for items marked as full-res from data attributes
|
||||
# (lazy-loaded images report thumbnail dimensions, not actual full-res dimensions)
|
||||
is_full_res = item.get('isFullRes', False)
|
||||
width = item.get('width', 0)
|
||||
height = item.get('height', 0)
|
||||
if width > 0 and height > 0:
|
||||
if width > 0 and height > 0 and not is_full_res:
|
||||
if width < self.min_width or height < self.min_height:
|
||||
continue
|
||||
|
||||
|
||||
@@ -0,0 +1,879 @@
|
||||
"""
|
||||
ModelMayhem Handler
|
||||
|
||||
Description: Handler for ModelMayhem.com portfolio/photo galleries
|
||||
Author: Eric Hiss (GitHub: EricRollei)
|
||||
Contact: eric@historic.camera, eric@rollei.us
|
||||
License: Dual License (Non-Commercial and Commercial Use)
|
||||
Copyright (c) 2025 Eric Hiss. All rights reserved.
|
||||
|
||||
Dual License:
|
||||
1. Non-Commercial Use: This software is licensed under the terms of the
|
||||
Creative Commons Attribution-NonCommercial 4.0 International License.
|
||||
To view a copy of this license, visit http://creativecommons.org/licenses/by-nc/4.0/
|
||||
|
||||
2. Commercial Use: For commercial use, a separate license is required.
|
||||
Please contact Eric Hiss at eric@historic.camera or eric@rollei.us for licensing options.
|
||||
|
||||
Dependencies:
|
||||
This code depends on several third-party libraries, each with its own license.
|
||||
See CREDITS.md for a comprehensive list of dependencies and their licenses.
|
||||
|
||||
Third-party code:
|
||||
- See CREDITS.md for complete list of dependencies
|
||||
"""
|
||||
|
||||
"""
|
||||
ModelMayhem-specific handler for the Web Image Scraper
|
||||
Extracts full-resolution images from ModelMayhem portfolios
|
||||
|
||||
URL patterns:
|
||||
- Profile: https://www.modelmayhem.com/4554791
|
||||
- Portfolio (all): https://www.modelmayhem.com/portfolio/4554791/viewall
|
||||
- Single photo: https://www.modelmayhem.com/portfolio/pic/48745727
|
||||
- Photo CDN: https://photos.modelmayhem.com/photos/251111/17/6913e94c8f27f.jpg
|
||||
"""
|
||||
|
||||
from site_handlers.base_handler import BaseSiteHandler
|
||||
from urllib.parse import urlparse, urljoin
|
||||
from typing import List, Dict, Any, Optional
|
||||
import re
|
||||
import json
|
||||
import time
|
||||
import asyncio
|
||||
import traceback
|
||||
|
||||
# Playwright (async) – load only if available
|
||||
try:
|
||||
from playwright.async_api import Page as AsyncPage
|
||||
from playwright.async_api import Browser, BrowserContext
|
||||
PLAYWRIGHT_AVAILABLE = True
|
||||
except ImportError:
|
||||
AsyncPage = None
|
||||
Browser = None
|
||||
BrowserContext = None
|
||||
PLAYWRIGHT_AVAILABLE = False
|
||||
|
||||
|
||||
class ModelMayhemHandler(BaseSiteHandler):
|
||||
"""Handler for ModelMayhem.com portfolio sites"""
|
||||
|
||||
# Class attributes for configuration
|
||||
PRIORITY = 40 # Higher priority for this specific handler
|
||||
|
||||
# Domains associated with ModelMayhem
|
||||
DOMAINS = [
|
||||
"modelmayhem.com",
|
||||
"www.modelmayhem.com",
|
||||
"m.modelmayhem.com",
|
||||
"secure.modelmayhem.com",
|
||||
"m.secure.modelmayhem.com",
|
||||
"photos.modelmayhem.com"
|
||||
]
|
||||
|
||||
@classmethod
|
||||
def can_handle(cls, url: str) -> bool:
|
||||
"""Check if this handler can process the URL"""
|
||||
url_lower = url.lower()
|
||||
for domain in cls.DOMAINS:
|
||||
if domain in url_lower:
|
||||
return True
|
||||
return False
|
||||
|
||||
def __init__(self, url, scraper=None):
|
||||
super().__init__(url, scraper)
|
||||
# Configuration defaults
|
||||
self.max_scroll_count = 15
|
||||
self.scroll_delay_ms = 1500
|
||||
self.use_stealth_mode = True
|
||||
self.retry_attempts = 2
|
||||
self.dynamic_content_wait_ms = 2000
|
||||
self.request_delay_ms = 1000
|
||||
self.last_request_time = 0
|
||||
|
||||
# Track state
|
||||
self.is_logged_in = False
|
||||
self.extracted_media_cache = {}
|
||||
self.seen_urls = set()
|
||||
self._username_to_resolve = None # For username-based URLs that need numeric ID resolution
|
||||
|
||||
# Load credentials from auth config
|
||||
self._load_api_credentials()
|
||||
|
||||
print(f"ModelMayhemHandler initialized for URL: {url}")
|
||||
|
||||
def get_trusted_domains(self) -> List[str]:
|
||||
"""Return list of trusted CDN domains for ModelMayhem"""
|
||||
return [
|
||||
"photos.modelmayhem.com", # Main photo CDN
|
||||
"assets.modelmayhem.com", # Assets CDN
|
||||
"modelmayhem.com", # Main domain
|
||||
"cloudfront.net" # Potential CDN
|
||||
]
|
||||
|
||||
def _load_api_credentials(self):
|
||||
"""Load credentials from the auth_config if available"""
|
||||
self.username = None
|
||||
self.password = None
|
||||
|
||||
if not hasattr(self, 'scraper') or not self.scraper:
|
||||
return
|
||||
|
||||
if not hasattr(self.scraper, 'auth_config') or not self.scraper.auth_config:
|
||||
return
|
||||
|
||||
auth_config = self.scraper.auth_config
|
||||
|
||||
# Try multiple domain variations
|
||||
domains_to_try = [
|
||||
"modelmayhem.com",
|
||||
"www.modelmayhem.com",
|
||||
"secure.modelmayhem.com"
|
||||
]
|
||||
|
||||
domain_config = None
|
||||
|
||||
# Look in 'sites' section first
|
||||
if 'sites' in auth_config:
|
||||
for domain in domains_to_try:
|
||||
if domain in auth_config['sites']:
|
||||
domain_config = auth_config['sites'][domain]
|
||||
print(f"ModelMayhemHandler: Found auth config for {domain}")
|
||||
break
|
||||
|
||||
# Also try directly in the config (older format)
|
||||
if not domain_config:
|
||||
for domain in domains_to_try:
|
||||
if domain in auth_config:
|
||||
domain_config = auth_config[domain]
|
||||
print(f"ModelMayhemHandler: Found auth config (legacy) for {domain}")
|
||||
break
|
||||
|
||||
if domain_config:
|
||||
self.username = domain_config.get('username')
|
||||
self.password = domain_config.get('password')
|
||||
# Load additional config if available
|
||||
if 'max_scroll_count' in domain_config:
|
||||
self.max_scroll_count = domain_config.get('max_scroll_count')
|
||||
if 'scroll_delay_ms' in domain_config:
|
||||
self.scroll_delay_ms = domain_config.get('scroll_delay_ms')
|
||||
if self.username:
|
||||
print(f"ModelMayhemHandler: Loaded credentials for user {self.username}")
|
||||
|
||||
def prefers_api(self) -> bool:
|
||||
"""This handler doesn't use APIs, it scrapes the web pages"""
|
||||
return False
|
||||
|
||||
def requires_api(self) -> bool:
|
||||
"""This handler doesn't require API access"""
|
||||
return False
|
||||
|
||||
async def _perform_login(self, page) -> bool:
|
||||
"""Perform login to ModelMayhem if credentials are available"""
|
||||
if not self.username or not self.password:
|
||||
print("ModelMayhemHandler: No credentials available, continuing without login")
|
||||
return False
|
||||
|
||||
if self.is_logged_in:
|
||||
return True
|
||||
|
||||
try:
|
||||
print(f"ModelMayhemHandler: Attempting login as {self.username}")
|
||||
|
||||
# Navigate to login page - use domcontentloaded for faster loading
|
||||
await page.goto("https://www.modelmayhem.com/login", wait_until="domcontentloaded", timeout=60000)
|
||||
await asyncio.sleep(3)
|
||||
|
||||
# Check if already logged in by looking for user menu or login form
|
||||
is_logged_in_already = await page.evaluate('''
|
||||
() => {
|
||||
// Check for signs of being logged in
|
||||
const userMenu = document.querySelector('.user-menu, .logged-in, [data-user], .account-menu');
|
||||
const loginForm = document.querySelector('form[action*="login"], #login-form, .login-form');
|
||||
return userMenu !== null || loginForm === null;
|
||||
}
|
||||
''')
|
||||
|
||||
if is_logged_in_already:
|
||||
print("ModelMayhemHandler: Already logged in")
|
||||
self.is_logged_in = True
|
||||
return True
|
||||
|
||||
# Fill in login form
|
||||
# Try multiple selector patterns for username field
|
||||
username_selectors = [
|
||||
'input[name="username"]',
|
||||
'input[name="email"]',
|
||||
'input[type="email"]',
|
||||
'#username',
|
||||
'#email',
|
||||
'input[placeholder*="mail"]',
|
||||
'input[placeholder*="user"]'
|
||||
]
|
||||
|
||||
for selector in username_selectors:
|
||||
try:
|
||||
username_field = await page.query_selector(selector)
|
||||
if username_field:
|
||||
await username_field.fill(self.username)
|
||||
print(f"ModelMayhemHandler: Filled username field using selector: {selector}")
|
||||
break
|
||||
except:
|
||||
continue
|
||||
|
||||
# Try multiple selector patterns for password field
|
||||
password_selectors = [
|
||||
'input[name="password"]',
|
||||
'input[type="password"]',
|
||||
'#password'
|
||||
]
|
||||
|
||||
for selector in password_selectors:
|
||||
try:
|
||||
password_field = await page.query_selector(selector)
|
||||
if password_field:
|
||||
await password_field.fill(self.password)
|
||||
print(f"ModelMayhemHandler: Filled password field using selector: {selector}")
|
||||
break
|
||||
except:
|
||||
continue
|
||||
|
||||
# Submit login form
|
||||
submit_selectors = [
|
||||
'button[type="submit"]',
|
||||
'input[type="submit"]',
|
||||
'button:has-text("Log In")',
|
||||
'button:has-text("Sign In")',
|
||||
'.login-button',
|
||||
'#login-submit'
|
||||
]
|
||||
|
||||
for selector in submit_selectors:
|
||||
try:
|
||||
submit_button = await page.query_selector(selector)
|
||||
if submit_button:
|
||||
await submit_button.click()
|
||||
print(f"ModelMayhemHandler: Clicked submit using selector: {selector}")
|
||||
break
|
||||
except:
|
||||
continue
|
||||
|
||||
# Wait for navigation/login to complete
|
||||
await asyncio.sleep(3)
|
||||
|
||||
# Verify login succeeded
|
||||
current_url = page.url
|
||||
if "login" not in current_url.lower():
|
||||
print("ModelMayhemHandler: Login successful")
|
||||
self.is_logged_in = True
|
||||
return True
|
||||
else:
|
||||
print("ModelMayhemHandler: Login may have failed, continuing anyway")
|
||||
return False
|
||||
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Login error: {e}")
|
||||
traceback.print_exc()
|
||||
return False
|
||||
|
||||
def _normalize_portfolio_url(self, url: str) -> str:
|
||||
"""Convert any ModelMayhem URL to the portfolio viewall URL"""
|
||||
# Extract user ID or username from various URL patterns
|
||||
patterns = [
|
||||
# Numeric ID patterns
|
||||
(r'modelmayhem\.com/portfolio/(\d+)/viewall', r'\1', True), # Already viewall: /portfolio/4554791/viewall
|
||||
(r'modelmayhem\.com/portfolio/(\d+)', r'\1', True), # Portfolio URL: /portfolio/4554791
|
||||
(r'modelmayhem\.com/(\d+)', r'\1', True), # Profile URL with ID: /4554791
|
||||
# Username patterns (alphanumeric, may include underscores)
|
||||
(r'modelmayhem\.com/portfolio/([a-zA-Z][a-zA-Z0-9_]+)/viewall', r'\1', False), # Already viewall with username
|
||||
(r'modelmayhem\.com/portfolio/([a-zA-Z][a-zA-Z0-9_]+)', r'\1', False), # Portfolio with username
|
||||
(r'modelmayhem\.com/([a-zA-Z][a-zA-Z0-9_]+)(?:/|$)', r'\1', False), # Profile URL with username: /albertobevacqua
|
||||
]
|
||||
|
||||
for pattern, group, is_numeric in patterns:
|
||||
match = re.search(pattern, url)
|
||||
if match:
|
||||
user_id = match.group(1)
|
||||
# Skip if it matches a reserved path
|
||||
if user_id.lower() in ['portfolio', 'login', 'signup', 'search', 'browse', 'help', 'about', 'contact', 'terms', 'privacy', 'pic']:
|
||||
continue
|
||||
|
||||
# For username-based URLs, we need to visit the profile page first to get the numeric ID
|
||||
# Store the username for later resolution
|
||||
if not is_numeric:
|
||||
self._username_to_resolve = user_id
|
||||
# Return profile URL - we'll resolve to numeric ID in extract_with_direct_playwright
|
||||
result_url = f"https://www.modelmayhem.com/{user_id}"
|
||||
print(f"ModelMayhemHandler: Username URL detected, will resolve numeric ID from profile: {result_url}")
|
||||
return result_url
|
||||
|
||||
result_url = f"https://www.modelmayhem.com/portfolio/{user_id}/viewall"
|
||||
print(f"ModelMayhemHandler: Normalized URL to {result_url}")
|
||||
return result_url
|
||||
|
||||
# If it's already a pic URL, just return it
|
||||
if '/portfolio/pic/' in url:
|
||||
return url
|
||||
|
||||
# Last resort: return the URL with /viewall appended if it looks like a profile
|
||||
print(f"ModelMayhemHandler: Could not normalize URL, using as-is: {url}")
|
||||
return url
|
||||
|
||||
def _upgrade_thumbnail_to_full_res(self, url: str) -> str:
|
||||
"""
|
||||
Convert a ModelMayhem thumbnail URL to full resolution.
|
||||
|
||||
ModelMayhem URL patterns:
|
||||
- Thumbnail: https://photos.modelmayhem.com/photos/251111/17/6913e94c8f27f_m.jpg
|
||||
- Full res: https://photos.modelmayhem.com/photos/251111/17/6913e94c8f27f.jpg
|
||||
|
||||
The _m suffix indicates medium thumbnail. Remove it for full resolution.
|
||||
"""
|
||||
if not url:
|
||||
return url
|
||||
|
||||
# Remove thumbnail suffixes (_m, _s, _t) before the extension
|
||||
# Pattern: filename_X.ext -> filename.ext where X is m, s, or t
|
||||
upgraded = re.sub(r'_[mst]\.([a-zA-Z]+)$', r'.\1', url)
|
||||
|
||||
if upgraded != url:
|
||||
print(f"ModelMayhemHandler: Upgraded URL: {url} -> {upgraded}")
|
||||
|
||||
return upgraded
|
||||
|
||||
async def _verify_and_get_best_url(self, page, thumbnail_url: str, verbose: bool = False) -> tuple[str, str]:
|
||||
"""
|
||||
Verify if the full-res URL exists, fall back to thumbnail if not.
|
||||
|
||||
Returns:
|
||||
tuple: (best_url, original_thumbnail_or_none)
|
||||
- best_url: The URL that should be used for download
|
||||
- original_thumbnail_or_none: The thumbnail URL if different from best_url, else None
|
||||
"""
|
||||
full_res_url = self._upgrade_thumbnail_to_full_res(thumbnail_url)
|
||||
|
||||
# If no upgrade happened (URL stayed the same), just use it
|
||||
if full_res_url == thumbnail_url:
|
||||
return (thumbnail_url, None)
|
||||
|
||||
# Verify the full-res URL exists
|
||||
try:
|
||||
response = await page.context.request.head(full_res_url, headers={
|
||||
'Referer': 'https://www.modelmayhem.com/',
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/123.0.0.0 Safari/537.36',
|
||||
})
|
||||
|
||||
if response.status < 400:
|
||||
# Full-res URL exists, use it with thumbnail as fallback
|
||||
if verbose:
|
||||
print(f"ModelMayhemHandler: Full-res URL verified: {full_res_url}")
|
||||
return (full_res_url, thumbnail_url)
|
||||
else:
|
||||
# Full-res URL doesn't exist, use thumbnail instead
|
||||
if verbose:
|
||||
print(f"ModelMayhemHandler: Full-res URL returned {response.status}, falling back to thumbnail: {thumbnail_url}")
|
||||
return (thumbnail_url, None)
|
||||
|
||||
except Exception as e:
|
||||
# On error, use thumbnail as safe fallback
|
||||
if verbose:
|
||||
print(f"ModelMayhemHandler: Error verifying full-res URL ({e}), using thumbnail: {thumbnail_url}")
|
||||
return (thumbnail_url, None)
|
||||
|
||||
async def extract_with_direct_playwright(self, page, **kwargs) -> List[Dict[str, Any]]:
|
||||
"""Main extraction method using Playwright"""
|
||||
print(f"ModelMayhemHandler: Starting extraction from {self.url}")
|
||||
start_time = time.time()
|
||||
media_items = []
|
||||
|
||||
try:
|
||||
# First, normalize the URL and check if we need to resolve a username
|
||||
target_url = self._normalize_portfolio_url(self.url)
|
||||
|
||||
# Attempt login first if we have credentials
|
||||
await self._perform_login(page)
|
||||
|
||||
# If we have a username to resolve, go to profile page first to get numeric ID
|
||||
if self._username_to_resolve:
|
||||
print(f"ModelMayhemHandler: Navigating to profile page to resolve username: {target_url}")
|
||||
try:
|
||||
# Use domcontentloaded instead of networkidle for faster loading
|
||||
await page.goto(target_url, wait_until="domcontentloaded", timeout=60000)
|
||||
await asyncio.sleep(3) # Give extra time for dynamic content
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Navigation timeout, retrying with load event: {e}")
|
||||
await page.goto(target_url, wait_until="load", timeout=60000)
|
||||
await asyncio.sleep(3)
|
||||
|
||||
numeric_id = await self._resolve_numeric_id_from_profile(page)
|
||||
if numeric_id:
|
||||
target_url = f"https://www.modelmayhem.com/portfolio/{numeric_id}/viewall"
|
||||
print(f"ModelMayhemHandler: Resolved username '{self._username_to_resolve}' to numeric ID {numeric_id}")
|
||||
else:
|
||||
print(f"ModelMayhemHandler: Could not resolve numeric ID for username '{self._username_to_resolve}'")
|
||||
return media_items
|
||||
|
||||
# Navigate to the portfolio page
|
||||
print(f"ModelMayhemHandler: Navigating to portfolio: {target_url}")
|
||||
try:
|
||||
await page.goto(target_url, wait_until="domcontentloaded", timeout=60000)
|
||||
await asyncio.sleep(3)
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Navigation timeout, retrying with load event: {e}")
|
||||
await page.goto(target_url, wait_until="load", timeout=60000)
|
||||
await asyncio.sleep(3)
|
||||
|
||||
# Disable worksafe mode if it's enabled (critical for seeing actual images)
|
||||
await self._disable_worksafe_mode(page)
|
||||
|
||||
# Scroll to load all images (lazy loading)
|
||||
await self._scroll_to_load_all(page)
|
||||
|
||||
# Check if this is a single pic page or portfolio page
|
||||
if '/portfolio/pic/' in self.url:
|
||||
# Single photo page - extract that one image
|
||||
media_items = await self._extract_single_photo(page)
|
||||
else:
|
||||
# Portfolio page - extract all photo links then get full-res images
|
||||
media_items = await self._extract_portfolio_images(page)
|
||||
|
||||
print(f"ModelMayhemHandler: Extracted {len(media_items)} images in {time.time() - start_time:.2f}s")
|
||||
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Error during extraction: {e}")
|
||||
traceback.print_exc()
|
||||
|
||||
return media_items
|
||||
|
||||
async def _resolve_numeric_id_from_profile(self, page) -> Optional[str]:
|
||||
"""
|
||||
Extract the numeric Model Mayhem ID from a profile page.
|
||||
The ID is shown in "Model Mayhem #: XXXXXX" or in portfolio link URLs.
|
||||
"""
|
||||
try:
|
||||
numeric_id = await page.evaluate('''
|
||||
() => {
|
||||
// Method 1: Look for "Model Mayhem #:" text
|
||||
const textContent = document.body.innerText;
|
||||
const mmMatch = textContent.match(/Model Mayhem #:\\s*(\\d+)/);
|
||||
if (mmMatch) return mmMatch[1];
|
||||
|
||||
// Method 2: Look for portfolio link with numeric ID
|
||||
const portfolioLink = document.querySelector('a[href*="/portfolio/"][href*="/viewall"]');
|
||||
if (portfolioLink) {
|
||||
const match = portfolioLink.href.match(/\\/portfolio\\/(\\d+)/);
|
||||
if (match) return match[1];
|
||||
}
|
||||
|
||||
// Method 3: Look for any link with /portfolio/NUMBER pattern
|
||||
const allLinks = document.querySelectorAll('a[href*="/portfolio/"]');
|
||||
for (const link of allLinks) {
|
||||
const match = link.href.match(/\\/portfolio\\/(\\d+)/);
|
||||
if (match) return match[1];
|
||||
}
|
||||
|
||||
// Method 4: Look for profile link in navigation
|
||||
const profileLinks = document.querySelectorAll('a[href^="/"]');
|
||||
for (const link of profileLinks) {
|
||||
const match = link.href.match(/modelmayhem\\.com\\/(\\d+)$/);
|
||||
if (match) return match[1];
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
''')
|
||||
|
||||
if numeric_id:
|
||||
print(f"ModelMayhemHandler: Found numeric ID: {numeric_id}")
|
||||
return numeric_id
|
||||
else:
|
||||
print("ModelMayhemHandler: Could not find numeric ID on profile page")
|
||||
return None
|
||||
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Error resolving numeric ID: {e}")
|
||||
return None
|
||||
|
||||
async def _disable_worksafe_mode(self, page):
|
||||
"""
|
||||
Disable worksafe mode to show actual images instead of placeholders.
|
||||
ModelMayhem defaults to worksafe mode which shows 'nopic_worksafe-on.gif' placeholders.
|
||||
"""
|
||||
try:
|
||||
# Check if worksafe mode is currently ON (link to turn it OFF is visible)
|
||||
worksafe_off_link = await page.query_selector('a[href="/worksafe/0"], a[href*="worksafe/0"]')
|
||||
|
||||
if worksafe_off_link:
|
||||
print("ModelMayhemHandler: Worksafe mode is ON, disabling it...")
|
||||
await worksafe_off_link.click()
|
||||
await asyncio.sleep(2) # Wait for page to reload/update
|
||||
print("ModelMayhemHandler: Worksafe mode disabled")
|
||||
else:
|
||||
# Check the toggle text to see current state
|
||||
page_text = await page.evaluate('() => document.body.innerText')
|
||||
if 'Toggle Worksafe Mode: Off' in page_text:
|
||||
print("ModelMayhemHandler: Worksafe mode already OFF")
|
||||
elif 'Toggle Worksafe Mode:' in page_text and 'On' in page_text:
|
||||
# Try clicking via JavaScript
|
||||
clicked = await page.evaluate('''
|
||||
() => {
|
||||
const links = document.querySelectorAll('a');
|
||||
for (const link of links) {
|
||||
if (link.textContent.trim() === 'Off' &&
|
||||
link.closest &&
|
||||
link.closest('[class*="worksafe"], [id*="worksafe"]') ||
|
||||
link.previousSibling?.textContent?.includes('Worksafe')) {
|
||||
link.click();
|
||||
return true;
|
||||
}
|
||||
}
|
||||
// Alternative: find by href pattern
|
||||
const wsLink = document.querySelector('a[href*="worksafe"]');
|
||||
if (wsLink && wsLink.textContent.trim() === 'Off') {
|
||||
wsLink.click();
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
''')
|
||||
if clicked:
|
||||
await asyncio.sleep(2)
|
||||
print("ModelMayhemHandler: Worksafe mode disabled via JS click")
|
||||
else:
|
||||
print("ModelMayhemHandler: Could not determine worksafe mode state")
|
||||
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Error disabling worksafe mode: {e}")
|
||||
|
||||
async def _scroll_to_load_all(self, page):
|
||||
"""Scroll down the page to trigger lazy loading of images"""
|
||||
print("ModelMayhemHandler: Scrolling to load all images...")
|
||||
|
||||
try:
|
||||
previous_height = 0
|
||||
scroll_count = 0
|
||||
|
||||
while scroll_count < self.max_scroll_count:
|
||||
# Get current scroll height
|
||||
current_height = await page.evaluate("document.body.scrollHeight")
|
||||
|
||||
if current_height == previous_height:
|
||||
# No new content loaded, we're done
|
||||
break
|
||||
|
||||
previous_height = current_height
|
||||
|
||||
# Scroll down
|
||||
await page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
|
||||
await asyncio.sleep(self.scroll_delay_ms / 1000)
|
||||
|
||||
scroll_count += 1
|
||||
print(f"ModelMayhemHandler: Scroll {scroll_count}/{self.max_scroll_count}")
|
||||
|
||||
# Scroll back to top
|
||||
await page.evaluate("window.scrollTo(0, 0)")
|
||||
await asyncio.sleep(0.5)
|
||||
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Scroll error: {e}")
|
||||
|
||||
async def _extract_single_photo(self, page) -> List[Dict[str, Any]]:
|
||||
"""Extract the full-resolution image from a single photo page"""
|
||||
media_items = []
|
||||
|
||||
try:
|
||||
# Look for the main photo on the page
|
||||
photo_data = await page.evaluate('''
|
||||
() => {
|
||||
const results = [];
|
||||
|
||||
// Look for images from photos.modelmayhem.com
|
||||
const imgs = document.querySelectorAll('img');
|
||||
for (const img of imgs) {
|
||||
const src = img.src || img.dataset.src || '';
|
||||
if (src.includes('photos.modelmayhem.com')) {
|
||||
results.push({
|
||||
url: src,
|
||||
width: img.naturalWidth || 0,
|
||||
height: img.naturalHeight || 0,
|
||||
alt: img.alt || ''
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Also look for og:image meta tag (often has full-res URL)
|
||||
const ogImage = document.querySelector('meta[property="og:image"]');
|
||||
if (ogImage && ogImage.content) {
|
||||
results.push({
|
||||
url: ogImage.content,
|
||||
width: 0,
|
||||
height: 0,
|
||||
alt: 'og:image'
|
||||
});
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
''')
|
||||
|
||||
for item in photo_data:
|
||||
url = item.get('url', '')
|
||||
if not url:
|
||||
continue
|
||||
|
||||
# Upgrade thumbnail to full resolution
|
||||
full_res_url = self._upgrade_thumbnail_to_full_res(url)
|
||||
|
||||
if full_res_url not in self.seen_urls:
|
||||
self.seen_urls.add(full_res_url)
|
||||
self.seen_urls.add(url) # Also mark original as seen
|
||||
media_items.append({
|
||||
'url': full_res_url,
|
||||
'type': 'image',
|
||||
'width': item.get('width', 0),
|
||||
'height': item.get('height', 0),
|
||||
'source_page': self.url,
|
||||
'trusted_cdn': True
|
||||
})
|
||||
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Error extracting single photo: {e}")
|
||||
|
||||
return media_items
|
||||
|
||||
async def _extract_portfolio_images(self, page) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Extract all images from a portfolio page.
|
||||
|
||||
Strategy: Extract thumbnail URLs from the portfolio grid page and upgrade them
|
||||
to full resolution URLs. This is much faster than visiting each individual
|
||||
pic page, and works because ModelMayhem uses predictable URL patterns:
|
||||
- Thumbnail: /photos/.../filename_m.jpg
|
||||
- Full res: /photos/.../filename.jpg
|
||||
"""
|
||||
media_items = []
|
||||
|
||||
try:
|
||||
# Debug: Log current URL and page title
|
||||
current_url = page.url
|
||||
page_title = await page.title()
|
||||
print(f"ModelMayhemHandler: Current page URL: {current_url}")
|
||||
print(f"ModelMayhemHandler: Page title: {page_title}")
|
||||
|
||||
# Debug: Count all images on page
|
||||
all_img_count = await page.evaluate('() => document.querySelectorAll("img").length')
|
||||
print(f"ModelMayhemHandler: Total img elements on page: {all_img_count}")
|
||||
|
||||
# Extract all images directly from the portfolio page
|
||||
# These are thumbnails but we'll upgrade them to full resolution
|
||||
page_images = await page.evaluate('''
|
||||
() => {
|
||||
const images = [];
|
||||
const debugInfo = { total: 0, modelmayhem: 0, worksafe: 0, other: [] };
|
||||
const imgs = document.querySelectorAll('img');
|
||||
debugInfo.total = imgs.length;
|
||||
|
||||
for (const img of imgs) {
|
||||
const src = img.src || img.dataset.src || img.getAttribute('data-src') || '';
|
||||
|
||||
// Skip worksafe placeholder images
|
||||
if (src.includes('nopic_worksafe') || src.includes('worksafe')) {
|
||||
debugInfo.worksafe++;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Only include photos.modelmayhem.com images (the actual photos)
|
||||
if (src.includes('photos.modelmayhem.com/photos/') ||
|
||||
src.includes('photos.modelmayhem.com/covers/')) {
|
||||
debugInfo.modelmayhem++;
|
||||
images.push({
|
||||
url: src,
|
||||
width: img.naturalWidth || 0,
|
||||
height: img.naturalHeight || 0
|
||||
});
|
||||
} else if (src && src.length > 10 && debugInfo.other.length < 5) {
|
||||
// Log first few non-matching URLs for debugging
|
||||
debugInfo.other.push(src.substring(0, 100));
|
||||
}
|
||||
}
|
||||
|
||||
// Also check for background images in divs (some galleries use this)
|
||||
const divs = document.querySelectorAll('[style*="background-image"]');
|
||||
for (const div of divs) {
|
||||
const style = div.getAttribute('style') || '';
|
||||
const match = style.match(/url\\(['"]?(https?:\\/\\/photos\\.modelmayhem\\.com[^'")\s]+)['"]?\\)/);
|
||||
if (match && match[1]) {
|
||||
debugInfo.modelmayhem++;
|
||||
images.push({
|
||||
url: match[1],
|
||||
width: 0,
|
||||
height: 0
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Check data attributes for lazy-loaded images
|
||||
const lazyImgs = document.querySelectorAll('[data-src*="photos.modelmayhem.com"], [data-lazy-src*="photos.modelmayhem.com"]');
|
||||
for (const img of lazyImgs) {
|
||||
const src = img.dataset.src || img.dataset.lazySrc || '';
|
||||
if (src && src.includes('photos.modelmayhem.com')) {
|
||||
debugInfo.modelmayhem++;
|
||||
images.push({
|
||||
url: src,
|
||||
width: 0,
|
||||
height: 0
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Return both images and debug info
|
||||
return { images: images, debug: debugInfo };
|
||||
}
|
||||
''')
|
||||
|
||||
# Extract debug info and images
|
||||
debug_info = page_images.get('debug', {})
|
||||
actual_images = page_images.get('images', [])
|
||||
|
||||
print(f"ModelMayhemHandler: Debug - Total imgs: {debug_info.get('total', 0)}, ModelMayhem imgs: {debug_info.get('modelmayhem', 0)}, Worksafe placeholders: {debug_info.get('worksafe', 0)}")
|
||||
if debug_info.get('other'):
|
||||
print(f"ModelMayhemHandler: Sample other image URLs: {debug_info.get('other', [])[:3]}")
|
||||
|
||||
print(f"ModelMayhemHandler: Found {len(actual_images)} ModelMayhem images on portfolio page")
|
||||
|
||||
# Upgrade each thumbnail URL to full resolution (with verification) and deduplicate
|
||||
print(f"ModelMayhemHandler: Verifying {len(actual_images)} image URLs...")
|
||||
verified_count = 0
|
||||
fallback_count = 0
|
||||
|
||||
for img in actual_images:
|
||||
thumb_url = img.get('url', '')
|
||||
if not thumb_url:
|
||||
continue
|
||||
|
||||
# Skip if we've already seen this thumbnail URL
|
||||
if thumb_url in self.seen_urls:
|
||||
continue
|
||||
|
||||
# Verify and get the best URL (full-res if it exists, else thumbnail)
|
||||
best_url, original_thumbnail = await self._verify_and_get_best_url(page, thumb_url)
|
||||
|
||||
# Skip if we've already seen the best URL
|
||||
if best_url in self.seen_urls:
|
||||
continue
|
||||
|
||||
self.seen_urls.add(best_url)
|
||||
self.seen_urls.add(thumb_url) # Also mark thumbnail as seen
|
||||
|
||||
# Track verification stats
|
||||
if original_thumbnail:
|
||||
verified_count += 1
|
||||
else:
|
||||
fallback_count += 1
|
||||
|
||||
media_items.append({
|
||||
'url': best_url,
|
||||
'type': 'image',
|
||||
'width': img.get('width', 0),
|
||||
'height': img.get('height', 0),
|
||||
'source_page': self.url,
|
||||
'trusted_cdn': True,
|
||||
'original_thumbnail': original_thumbnail
|
||||
})
|
||||
|
||||
print(f"ModelMayhemHandler: Extracted {len(media_items)} unique image URLs ({verified_count} full-res verified, {fallback_count} using thumbnail)")
|
||||
|
||||
# If we didn't find many images, fall back to visiting individual pic pages
|
||||
# This handles edge cases where the portfolio page doesn't show all thumbnails
|
||||
if len(media_items) < 10:
|
||||
print("ModelMayhemHandler: Few images found, trying pic page extraction as fallback...")
|
||||
fallback_items = await self._extract_via_pic_pages(page)
|
||||
|
||||
# Add any new URLs from fallback
|
||||
for item in fallback_items:
|
||||
url = item.get('url', '')
|
||||
if url and url not in self.seen_urls:
|
||||
self.seen_urls.add(url)
|
||||
media_items.append(item)
|
||||
|
||||
print(f"ModelMayhemHandler: After fallback, total {len(media_items)} images")
|
||||
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Error extracting portfolio images: {e}")
|
||||
traceback.print_exc()
|
||||
|
||||
return media_items
|
||||
|
||||
async def _extract_via_pic_pages(self, page) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Fallback method: Visit individual pic pages to extract full-res images.
|
||||
Used when the portfolio page doesn't have all thumbnails visible.
|
||||
"""
|
||||
media_items = []
|
||||
|
||||
try:
|
||||
# Collect all portfolio pic links
|
||||
pic_links = await page.evaluate('''
|
||||
() => {
|
||||
const links = [];
|
||||
const anchors = document.querySelectorAll('a[href*="/portfolio/pic/"]');
|
||||
for (const a of anchors) {
|
||||
if (a.href && !links.includes(a.href)) {
|
||||
links.push(a.href);
|
||||
}
|
||||
}
|
||||
return links;
|
||||
}
|
||||
''')
|
||||
|
||||
print(f"ModelMayhemHandler: Found {len(pic_links)} pic page links for fallback extraction")
|
||||
|
||||
# Visit each pic page (no artificial limit - get all images)
|
||||
for i, pic_link in enumerate(pic_links):
|
||||
try:
|
||||
if self.request_delay_ms > 0:
|
||||
await asyncio.sleep(self.request_delay_ms / 1000)
|
||||
|
||||
if (i + 1) % 10 == 0:
|
||||
print(f"ModelMayhemHandler: Processing pic {i+1}/{len(pic_links)}")
|
||||
|
||||
# Navigate to the pic page
|
||||
await page.goto(pic_link, wait_until="domcontentloaded", timeout=15000)
|
||||
await asyncio.sleep(0.5)
|
||||
|
||||
# Extract the full-res image from this page
|
||||
pic_images = await self._extract_single_photo(page)
|
||||
|
||||
for item in pic_images:
|
||||
url = item.get('url', '')
|
||||
if url and url not in self.seen_urls:
|
||||
self.seen_urls.add(url)
|
||||
media_items.append(item)
|
||||
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Error processing pic {i+1}: {e}")
|
||||
continue
|
||||
|
||||
except Exception as e:
|
||||
print(f"ModelMayhemHandler: Error in pic page extraction: {e}")
|
||||
|
||||
return media_items
|
||||
|
||||
async def extract_media_items_async(self, page_adaptor) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Async extraction method - wrapper that gets Playwright page from adaptor
|
||||
"""
|
||||
print(f"ModelMayhemHandler: extract_media_items_async called")
|
||||
|
||||
# Try to get the underlying Playwright page
|
||||
pw_page = None
|
||||
if hasattr(page_adaptor, 'page'):
|
||||
pw_page = page_adaptor.page
|
||||
elif hasattr(self, 'get_playwright_page'):
|
||||
pw_page = await self.get_playwright_page(page_adaptor)
|
||||
|
||||
if pw_page:
|
||||
return await self.extract_with_direct_playwright(pw_page)
|
||||
else:
|
||||
print("ModelMayhemHandler: No Playwright page available")
|
||||
return []
|
||||
@@ -0,0 +1,402 @@
|
||||
"""
|
||||
Portfolio Handler
|
||||
|
||||
Description: Handler for model portfolio sites like juliaromanova.com that use clickable image galleries
|
||||
Author: Eric Hiss (GitHub: EricRollei)
|
||||
Contact: eric@historic.camera, eric@rollei.us
|
||||
License: Dual License (Non-Commercial and Commercial Use)
|
||||
Copyright (c) 2025 Eric Hiss. All rights reserved.
|
||||
|
||||
Dual License:
|
||||
1. Non-Commercial Use: This software is licensed under the terms of the
|
||||
Creative Commons Attribution-NonCommercial 4.0 International License.
|
||||
To view a copy of this license, visit http://creativecommons.org/licenses/by-nc/4.0/
|
||||
|
||||
2. Commercial Use: For commercial use, a separate license is required.
|
||||
Please contact Eric Hiss at eric@historic.camera or eric@rollei.us for licensing options.
|
||||
"""
|
||||
|
||||
"""
|
||||
Portfolio handler for simple model portfolio sites with clickable image galleries.
|
||||
Handles sites like juliaromanova.com, artfolio-powered sites, and similar simple galleries.
|
||||
"""
|
||||
|
||||
from site_handlers.base_handler import BaseSiteHandler
|
||||
from urllib.parse import urljoin, urlparse
|
||||
import re
|
||||
import time
|
||||
import traceback
|
||||
|
||||
# Try importing Playwright types safely
|
||||
try:
|
||||
from playwright.async_api import Page as AsyncPage
|
||||
PLAYWRIGHT_AVAILABLE = True
|
||||
except ImportError:
|
||||
AsyncPage = None
|
||||
PLAYWRIGHT_AVAILABLE = False
|
||||
|
||||
|
||||
class PortfolioHandler(BaseSiteHandler):
|
||||
"""
|
||||
Handler for simple portfolio sites with clickable image galleries.
|
||||
|
||||
Targets sites that:
|
||||
- Have thumbnail galleries where clicking opens larger images
|
||||
- Use patterns like /galleries/, /portfolio/, /photos/
|
||||
- Often use numbered image paths or ID-based URLs
|
||||
- Include artfolio.com-powered sites
|
||||
"""
|
||||
|
||||
# Sites and patterns this handler can process
|
||||
KNOWN_PORTFOLIO_PATTERNS = [
|
||||
r'juliaromanova\.com',
|
||||
r'artfolio\.com',
|
||||
r'/galleries?/',
|
||||
r'/portfolio/',
|
||||
r'/photos?/',
|
||||
r'/book/',
|
||||
r'/albums?/',
|
||||
r'/models?/',
|
||||
r'/shoots?/',
|
||||
]
|
||||
|
||||
# Image URL patterns that indicate resizable images (can be upgraded to larger versions)
|
||||
RESIZABLE_PATTERNS = [
|
||||
(r'g_10_', 'g_30_'), # Artfolio pattern: 10% thumb to 30% large
|
||||
(r'_thumb\.', '_full.'),
|
||||
(r'_small\.', '_large.'),
|
||||
(r'_t\.', '_l.'),
|
||||
(r'/s/', '/l/'), # some sites use /s/ for small
|
||||
(r'/small/', '/large/'),
|
||||
(r'/thumb/', '/full/'),
|
||||
(r'width=\d+', 'width=2000'),
|
||||
(r'w=\d+', 'w=2000'),
|
||||
(r'size=\w+', 'size=original'),
|
||||
]
|
||||
|
||||
# Sites that have their own specific handlers - don't handle these
|
||||
EXCLUDED_DOMAINS = [
|
||||
'modelmayhem.com',
|
||||
'instagram.com',
|
||||
'flickr.com',
|
||||
'500px.com',
|
||||
'deviantart.com',
|
||||
'artstation.com',
|
||||
'behance.net',
|
||||
]
|
||||
|
||||
@classmethod
|
||||
def can_handle(cls, url):
|
||||
"""Check if this handler can process the URL"""
|
||||
url_lower = url.lower()
|
||||
|
||||
# Don't handle sites that have their own specific handlers
|
||||
for excluded_domain in cls.EXCLUDED_DOMAINS:
|
||||
if excluded_domain in url_lower:
|
||||
return False
|
||||
|
||||
# Check for known portfolio patterns
|
||||
for pattern in cls.KNOWN_PORTFOLIO_PATTERNS:
|
||||
if re.search(pattern, url_lower, re.IGNORECASE):
|
||||
print(f"PortfolioHandler can handle: {url} (matched pattern: {pattern})")
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def __init__(self, url, scraper=None):
|
||||
"""Initialize the portfolio handler"""
|
||||
super().__init__(url, scraper)
|
||||
self.debug_mode = True
|
||||
self.gallery_links = []
|
||||
self.seen_urls = set()
|
||||
self.image_extensions = ['.jpg', '.jpeg', '.png', '.gif', '.webp', '.bmp', '.tiff']
|
||||
print(f"PortfolioHandler initialized for URL: {url}")
|
||||
|
||||
def prefers_api(self):
|
||||
"""This handler doesn't use APIs"""
|
||||
return False
|
||||
|
||||
def requires_api(self):
|
||||
"""This handler doesn't require API access"""
|
||||
return False
|
||||
|
||||
async def extract_media_items_async(self, page):
|
||||
"""
|
||||
Extract media items from portfolio gallery pages.
|
||||
|
||||
This handler:
|
||||
1. Finds all gallery links (clickable thumbnails)
|
||||
2. Extracts direct image URLs from thumbnails
|
||||
3. Upgrades thumbnail URLs to full-size versions
|
||||
"""
|
||||
print(f"PortfolioHandler: Extracting media from {self.url}")
|
||||
start_time = time.time()
|
||||
media_items = []
|
||||
|
||||
# Get the Playwright page
|
||||
pw_page = await self._get_playwright_page_async(page)
|
||||
if not pw_page:
|
||||
print("No Playwright page available, falling back to DOM extraction")
|
||||
return await self._extract_from_html(page)
|
||||
|
||||
try:
|
||||
# Wait for the page to load
|
||||
await pw_page.wait_for_load_state('networkidle', timeout=15000)
|
||||
|
||||
# First, try to extract all images directly with full resolution
|
||||
media_items = await self._extract_and_upgrade_images(pw_page)
|
||||
|
||||
# If we found images, return them
|
||||
if media_items:
|
||||
print(f"PortfolioHandler: Found {len(media_items)} media items")
|
||||
return media_items
|
||||
|
||||
# Fallback: Extract from gallery links and navigate to each
|
||||
gallery_items = await self._extract_gallery_links(pw_page)
|
||||
if gallery_items:
|
||||
media_items.extend(gallery_items)
|
||||
|
||||
except Exception as e:
|
||||
print(f"PortfolioHandler error: {e}")
|
||||
traceback.print_exc()
|
||||
|
||||
print(f"PortfolioHandler: Extracted {len(media_items)} media items in {time.time() - start_time:.2f}s")
|
||||
return media_items
|
||||
|
||||
async def _extract_and_upgrade_images(self, pw_page):
|
||||
"""Extract images from the page and upgrade to full resolution"""
|
||||
media_items = []
|
||||
|
||||
try:
|
||||
# Get all images from the page
|
||||
images = await pw_page.evaluate('''
|
||||
() => {
|
||||
const imgs = document.querySelectorAll('img');
|
||||
const result = [];
|
||||
|
||||
imgs.forEach(img => {
|
||||
const src = img.src || img.dataset.src || img.getAttribute('data-lazy-src') || '';
|
||||
if (src && !src.includes('data:image') && !src.includes('blank.gif')) {
|
||||
// Get the largest version we can find
|
||||
const srcset = img.srcset || '';
|
||||
let largestSrc = src;
|
||||
let largestWidth = img.naturalWidth || 0;
|
||||
|
||||
// Parse srcset for larger versions
|
||||
if (srcset) {
|
||||
const srcsetParts = srcset.split(',');
|
||||
for (const part of srcsetParts) {
|
||||
const match = part.trim().match(/^(\S+)\s+(\d+)w$/);
|
||||
if (match) {
|
||||
const [, url, width] = match;
|
||||
if (parseInt(width) > largestWidth) {
|
||||
largestWidth = parseInt(width);
|
||||
largestSrc = url;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Get parent link if exists (for galleries)
|
||||
let parentLink = null;
|
||||
const link = img.closest('a');
|
||||
if (link && link.href) {
|
||||
parentLink = link.href;
|
||||
}
|
||||
|
||||
result.push({
|
||||
src: largestSrc,
|
||||
originalSrc: src,
|
||||
alt: img.alt || '',
|
||||
title: img.title || '',
|
||||
width: img.naturalWidth || 0,
|
||||
height: img.naturalHeight || 0,
|
||||
parentLink: parentLink
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
return result;
|
||||
}
|
||||
''')
|
||||
|
||||
print(f"PortfolioHandler: Found {len(images)} images on page")
|
||||
|
||||
for img_data in images:
|
||||
src = img_data.get('src', '')
|
||||
if not src:
|
||||
continue
|
||||
|
||||
# Skip tracking pixels and tiny images
|
||||
width = img_data.get('width', 0)
|
||||
height = img_data.get('height', 0)
|
||||
if width > 0 and width < 50 and height > 0 and height < 50:
|
||||
continue
|
||||
|
||||
# Try to upgrade to full resolution
|
||||
full_res_url = self._upgrade_to_full_resolution(src)
|
||||
|
||||
# Skip if we've already seen this URL
|
||||
if full_res_url in self.seen_urls:
|
||||
continue
|
||||
self.seen_urls.add(full_res_url)
|
||||
|
||||
media_item = {
|
||||
'url': full_res_url,
|
||||
'original_url': src,
|
||||
'type': 'image',
|
||||
'alt_text': img_data.get('alt', ''),
|
||||
'title': img_data.get('title', ''),
|
||||
'source': 'portfolio_handler',
|
||||
'parent_link': img_data.get('parentLink', '')
|
||||
}
|
||||
media_items.append(media_item)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error extracting images: {e}")
|
||||
traceback.print_exc()
|
||||
|
||||
return media_items
|
||||
|
||||
def _upgrade_to_full_resolution(self, url):
|
||||
"""
|
||||
Upgrade a thumbnail URL to full resolution version.
|
||||
Uses known patterns to transform URLs.
|
||||
"""
|
||||
upgraded_url = url
|
||||
|
||||
for pattern, replacement in self.RESIZABLE_PATTERNS:
|
||||
if re.search(pattern, url):
|
||||
upgraded_url = re.sub(pattern, replacement, url)
|
||||
if upgraded_url != url:
|
||||
print(f"Upgraded URL: {url[:50]}... -> {upgraded_url[:50]}...")
|
||||
break
|
||||
|
||||
return upgraded_url
|
||||
|
||||
async def _extract_gallery_links(self, pw_page):
|
||||
"""Extract links to gallery pages (for multi-page galleries)"""
|
||||
media_items = []
|
||||
|
||||
try:
|
||||
# Get all links that look like gallery item links
|
||||
links = await pw_page.evaluate('''
|
||||
() => {
|
||||
const links = document.querySelectorAll('a');
|
||||
const galleryLinks = [];
|
||||
|
||||
for (const link of links) {
|
||||
const href = link.href;
|
||||
// Look for gallery/portfolio-style links with numbers or slugs
|
||||
if (href && (
|
||||
href.match(/\\/galleries?\\/.*\\/\\d+/) ||
|
||||
href.match(/\\/portfolio\\/.*\\//) ||
|
||||
href.match(/\\/photos?\\/.*\\//) ||
|
||||
href.match(/\\/book\\/[\\w-]+/) ||
|
||||
href.match(/\\/albums?\\/.*\\//)
|
||||
)) {
|
||||
// Check if link contains an image (thumbnail)
|
||||
const img = link.querySelector('img');
|
||||
if (img) {
|
||||
galleryLinks.push({
|
||||
href: href,
|
||||
title: link.title || img.alt || '',
|
||||
thumbSrc: img.src || ''
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return galleryLinks;
|
||||
}
|
||||
''')
|
||||
|
||||
print(f"PortfolioHandler: Found {len(links)} gallery links")
|
||||
|
||||
# For each gallery link, try to extract the full-size image
|
||||
for link_data in links:
|
||||
href = link_data.get('href', '')
|
||||
thumb_src = link_data.get('thumbSrc', '')
|
||||
|
||||
if href in self.seen_urls:
|
||||
continue
|
||||
self.seen_urls.add(href)
|
||||
|
||||
# Try to derive full-res URL from thumbnail
|
||||
if thumb_src:
|
||||
full_res_url = self._upgrade_to_full_resolution(thumb_src)
|
||||
|
||||
if full_res_url not in self.seen_urls:
|
||||
self.seen_urls.add(full_res_url)
|
||||
media_item = {
|
||||
'url': full_res_url,
|
||||
'original_url': thumb_src,
|
||||
'type': 'image',
|
||||
'alt_text': link_data.get('title', ''),
|
||||
'source': 'portfolio_handler',
|
||||
'gallery_page': href
|
||||
}
|
||||
media_items.append(media_item)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error extracting gallery links: {e}")
|
||||
traceback.print_exc()
|
||||
|
||||
return media_items
|
||||
|
||||
async def _get_playwright_page_async(self, page):
|
||||
"""Get the underlying Playwright page object"""
|
||||
if not PLAYWRIGHT_AVAILABLE:
|
||||
return None
|
||||
|
||||
# Handle different page wrapper types
|
||||
if hasattr(page, '_playwright_page'):
|
||||
return page._playwright_page
|
||||
elif hasattr(page, 'page'):
|
||||
return page.page
|
||||
elif isinstance(page, AsyncPage) if AsyncPage else False:
|
||||
return page
|
||||
elif hasattr(page, 'playwright_page'):
|
||||
return page.playwright_page
|
||||
|
||||
return page
|
||||
|
||||
async def _extract_from_html(self, page):
|
||||
"""Fallback HTML extraction when Playwright isn't available"""
|
||||
media_items = []
|
||||
|
||||
try:
|
||||
# Try to get HTML content
|
||||
html = None
|
||||
if hasattr(page, 'html'):
|
||||
html = page.html
|
||||
elif hasattr(page, 'content'):
|
||||
html = await page.content() if callable(page.content) else page.content
|
||||
|
||||
if html:
|
||||
# Simple regex to find image URLs
|
||||
img_pattern = r'<img[^>]+src=["\']([^"\']+)["\']'
|
||||
matches = re.findall(img_pattern, html, re.IGNORECASE)
|
||||
|
||||
for src in matches:
|
||||
if src and not src.startswith('data:'):
|
||||
full_url = urljoin(self.url, src)
|
||||
full_res = self._upgrade_to_full_resolution(full_url)
|
||||
|
||||
if full_res not in self.seen_urls:
|
||||
self.seen_urls.add(full_res)
|
||||
media_items.append({
|
||||
'url': full_res,
|
||||
'original_url': full_url,
|
||||
'type': 'image',
|
||||
'source': 'portfolio_handler_html'
|
||||
})
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error in HTML extraction: {e}")
|
||||
|
||||
return media_items
|
||||
|
||||
async def extract_with_direct_playwright(self, page, **kwargs):
|
||||
"""Direct Playwright extraction method called by the scraper"""
|
||||
return await self.extract_media_items_async(page)
|
||||
+112
-12
@@ -123,14 +123,18 @@ class RedditHandler(BaseSiteHandler):
|
||||
self._extract_identifiers_from_url()
|
||||
|
||||
# --- Load API credentials ---
|
||||
# self._load_api_credentials() remove as main code has not loaded confgig yet
|
||||
# Credentials will be loaded lazily in prefers_api() when the scraper's auth_config is available
|
||||
|
||||
|
||||
def prefers_api(self) -> bool:
|
||||
"""Reddit handler prefers API if credentials were loaded."""
|
||||
# Ensure credentials are loaded from scraper's auth_config if not already done
|
||||
if not self.api_available and self.scraper:
|
||||
self._load_api_credentials()
|
||||
|
||||
# Check if the necessary attributes exist AND have truthy values
|
||||
has_creds = bool(self.client_id and self.client_secret)
|
||||
print(f"RedditHandler prefers_api check. Result: {has_creds}")
|
||||
print(f"RedditHandler prefers_api check. Result: {has_creds} (api_available={self.api_available})")
|
||||
return has_creds
|
||||
|
||||
async def extract_api_data_async(self, **kwargs) -> list:
|
||||
@@ -316,7 +320,24 @@ class RedditHandler(BaseSiteHandler):
|
||||
# Check if it's a self post (text post)
|
||||
if not submission.is_self and hasattr(submission, 'url'):
|
||||
url = submission.url
|
||||
if self._is_image_url(url):
|
||||
|
||||
# Check if this is an external video host that needs resolution
|
||||
if self._is_external_video_host(url):
|
||||
resolved = await self._resolve_external_video_url(url)
|
||||
if resolved:
|
||||
media_items.append({
|
||||
'url': resolved['url'],
|
||||
'alt': post_title,
|
||||
'title': resolved.get('title') or meta_title,
|
||||
'source_url': post_url,
|
||||
'credits': credit_line,
|
||||
'type': 'video',
|
||||
'width': resolved.get('width', 0),
|
||||
'height': resolved.get('height', 0),
|
||||
'duration': resolved.get('duration'),
|
||||
'_headers': {'Referer': url} # Use original URL as referer
|
||||
})
|
||||
elif self._is_image_url(url):
|
||||
media_items.append({
|
||||
'url': url,
|
||||
'alt': post_title,
|
||||
@@ -327,15 +348,29 @@ class RedditHandler(BaseSiteHandler):
|
||||
'_headers': {'Referer': post_url}
|
||||
})
|
||||
elif self._is_video_url(url):
|
||||
media_items.append({
|
||||
'url': url,
|
||||
'alt': post_title,
|
||||
'title': meta_title,
|
||||
'source_url': post_url,
|
||||
'credits': credit_line,
|
||||
'type': 'video',
|
||||
'_headers': {'Referer': post_url}
|
||||
})
|
||||
# For v.redd.it links, they usually need special handling too
|
||||
if 'v.redd.it' in url:
|
||||
resolved = await self._resolve_external_video_url(url)
|
||||
if resolved:
|
||||
media_items.append({
|
||||
'url': resolved['url'],
|
||||
'alt': post_title,
|
||||
'title': meta_title,
|
||||
'source_url': post_url,
|
||||
'credits': credit_line,
|
||||
'type': 'video',
|
||||
'_headers': {'Referer': post_url}
|
||||
})
|
||||
else:
|
||||
media_items.append({
|
||||
'url': url,
|
||||
'alt': post_title,
|
||||
'title': meta_title,
|
||||
'source_url': post_url,
|
||||
'credits': credit_line,
|
||||
'type': 'video',
|
||||
'_headers': {'Referer': post_url}
|
||||
})
|
||||
|
||||
# Handle Reddit-hosted videos
|
||||
if submission.is_video and submission.media and 'reddit_video' in submission.media:
|
||||
@@ -502,6 +537,71 @@ class RedditHandler(BaseSiteHandler):
|
||||
|
||||
return False
|
||||
|
||||
def _is_external_video_host(self, url: str) -> bool:
|
||||
"""Check if URL is from an external video host that needs resolution."""
|
||||
external_hosts = ['redgifs.com', 'gfycat.com', 'imgur.com/a/', 'imgur.com/gallery/']
|
||||
return any(host in url.lower() for host in external_hosts)
|
||||
|
||||
async def _resolve_external_video_url(self, url: str) -> Optional[Dict[str, Any]]:
|
||||
"""
|
||||
Resolve external video host URLs (RedGifs, Gfycat, Imgur) to direct video URLs.
|
||||
Uses yt-dlp as it has excellent support for these sites.
|
||||
|
||||
Returns:
|
||||
Dict with 'url' and 'type' keys, or None if resolution fails
|
||||
"""
|
||||
import subprocess
|
||||
import tempfile
|
||||
|
||||
try:
|
||||
print(f"RedditHandler: Resolving external video URL: {url}")
|
||||
|
||||
# Use yt-dlp to extract video info
|
||||
result = subprocess.run(
|
||||
['yt-dlp', '--dump-json', '--no-download', url],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=30
|
||||
)
|
||||
|
||||
if result.returncode == 0 and result.stdout:
|
||||
info = json.loads(result.stdout)
|
||||
|
||||
# Get the best video URL
|
||||
video_url = info.get('url')
|
||||
if not video_url:
|
||||
# Try to get from formats
|
||||
formats = info.get('formats', [])
|
||||
if formats:
|
||||
# Prefer mp4, then webm
|
||||
for fmt in reversed(formats):
|
||||
if fmt.get('ext') == 'mp4' and fmt.get('url'):
|
||||
video_url = fmt['url']
|
||||
break
|
||||
if not video_url:
|
||||
video_url = formats[-1].get('url')
|
||||
|
||||
if video_url:
|
||||
print(f"RedditHandler: Resolved to: {video_url[:80]}...")
|
||||
return {
|
||||
'url': video_url,
|
||||
'type': 'video',
|
||||
'ext': info.get('ext', 'mp4'),
|
||||
'title': info.get('title', ''),
|
||||
'duration': info.get('duration'),
|
||||
'width': info.get('width'),
|
||||
'height': info.get('height')
|
||||
}
|
||||
else:
|
||||
print(f"RedditHandler: yt-dlp failed for {url}: {result.stderr[:200] if result.stderr else 'no error'}")
|
||||
|
||||
except subprocess.TimeoutExpired:
|
||||
print(f"RedditHandler: Timeout resolving {url}")
|
||||
except Exception as e:
|
||||
print(f"RedditHandler: Error resolving external video: {e}")
|
||||
|
||||
return None
|
||||
|
||||
async def _extract_media_from_dom_async(self, page: AsyncPage, **kwargs) -> list:
|
||||
"""Extract media from DOM using async Playwright"""
|
||||
media_items = []
|
||||
|
||||
@@ -0,0 +1,695 @@
|
||||
"""
|
||||
Wix Handler
|
||||
|
||||
Description: Handler for Wix-powered websites with image galleries
|
||||
Author: Eric Hiss (GitHub: EricRollei)
|
||||
Contact: eric@historic.camera, eric@rollei.us
|
||||
License: Dual License (Non-Commercial and Commercial Use)
|
||||
Copyright (c) 2025 Eric Hiss. All rights reserved.
|
||||
|
||||
Dual License:
|
||||
1. Non-Commercial Use: This software is licensed under the terms of the
|
||||
Creative Commons Attribution-NonCommercial 4.0 International License.
|
||||
To view a copy of this license, visit http://creativecommons.org/licenses/by-nc/4.0/
|
||||
|
||||
2. Commercial Use: For commercial use, a separate license is required.
|
||||
Please contact Eric Hiss at eric@historic.camera or eric@rollei.us for licensing options.
|
||||
"""
|
||||
|
||||
"""
|
||||
Wix handler for websites built on the Wix platform.
|
||||
Handles image extraction and resolution upgrades for Wix's image CDN.
|
||||
|
||||
Wix image URL patterns:
|
||||
- static.wixstatic.com/media/[id]/v1/fill/w_[w],h_[h],q_[q],...
|
||||
- static.wixstatic.com/media/[id]~mv2.jpg (original)
|
||||
- static.wixstatic.com/media/[id].jpg
|
||||
|
||||
Transform parameters:
|
||||
- w_[width] - target width
|
||||
- h_[height] - target height
|
||||
- q_[quality] - quality (1-100)
|
||||
- al_c - alignment center
|
||||
- usm_0.66_1.00_0.01 - unsharp mask
|
||||
- enc_avif - encoding format (avif, webp, jpg, png)
|
||||
"""
|
||||
|
||||
from site_handlers.base_handler import BaseSiteHandler
|
||||
from urllib.parse import urljoin, urlparse, parse_qs, urlencode
|
||||
import re
|
||||
import time
|
||||
import traceback
|
||||
|
||||
# Try importing Playwright types safely
|
||||
try:
|
||||
from playwright.async_api import Page as AsyncPage
|
||||
PLAYWRIGHT_AVAILABLE = True
|
||||
except ImportError:
|
||||
AsyncPage = None
|
||||
PLAYWRIGHT_AVAILABLE = False
|
||||
|
||||
|
||||
class WixHandler(BaseSiteHandler):
|
||||
"""
|
||||
Handler for Wix-powered websites.
|
||||
|
||||
Wix sites use static.wixstatic.com for image hosting with URL-based
|
||||
image transformations. This handler extracts images and upgrades
|
||||
them to maximum resolution.
|
||||
"""
|
||||
|
||||
# Patterns to identify Wix sites
|
||||
WIX_PATTERNS = [
|
||||
r'wixstatic\.com',
|
||||
r'wix\.com',
|
||||
r'wixsite\.com',
|
||||
r'_wix_',
|
||||
r'wix-code',
|
||||
]
|
||||
|
||||
# Known Wix-powered domains (model agencies, portfolios, etc.)
|
||||
KNOWN_WIX_DOMAINS = [
|
||||
r'new1mgmt\.com',
|
||||
r'modelwerk\.de',
|
||||
r'nextmodels\.com',
|
||||
# Add more known Wix sites as discovered
|
||||
]
|
||||
|
||||
@classmethod
|
||||
def can_handle(cls, url):
|
||||
"""Check if this handler can process the URL"""
|
||||
url_lower = url.lower()
|
||||
|
||||
# Check for known Wix domains
|
||||
for pattern in cls.KNOWN_WIX_DOMAINS:
|
||||
if re.search(pattern, url_lower, re.IGNORECASE):
|
||||
print(f"WixHandler can handle: {url} (known Wix domain: {pattern})")
|
||||
return True
|
||||
|
||||
# Check for Wix patterns in URL
|
||||
for pattern in cls.WIX_PATTERNS:
|
||||
if re.search(pattern, url_lower, re.IGNORECASE):
|
||||
print(f"WixHandler can handle: {url} (matched pattern: {pattern})")
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def __init__(self, url, scraper=None):
|
||||
"""Initialize the Wix handler"""
|
||||
super().__init__(url, scraper)
|
||||
self.debug_mode = True
|
||||
self.seen_urls = set()
|
||||
self.seen_base_ids = set() # Track base image IDs to avoid duplicates
|
||||
print(f"WixHandler initialized for URL: {url}")
|
||||
|
||||
def prefers_api(self):
|
||||
"""This handler doesn't use APIs"""
|
||||
return False
|
||||
|
||||
def requires_api(self):
|
||||
"""This handler doesn't require API access"""
|
||||
return False
|
||||
|
||||
async def extract_media_items_async(self, page):
|
||||
"""
|
||||
Extract media items from Wix-powered pages.
|
||||
"""
|
||||
print(f"WixHandler: Extracting media from {self.url}")
|
||||
start_time = time.time()
|
||||
media_items = []
|
||||
|
||||
# Get the Playwright page
|
||||
pw_page = await self._get_playwright_page_async(page)
|
||||
if not pw_page:
|
||||
print("No Playwright page available")
|
||||
return []
|
||||
|
||||
try:
|
||||
# Wait for the page to fully load
|
||||
await pw_page.wait_for_load_state('networkidle', timeout=20000)
|
||||
|
||||
# Give extra time for Wix's lazy loading
|
||||
await pw_page.wait_for_timeout(2000)
|
||||
|
||||
# Scroll to trigger lazy loading
|
||||
await self._scroll_page(pw_page)
|
||||
|
||||
# Extract images from multiple sources
|
||||
media_items = await self._extract_wix_images(pw_page)
|
||||
|
||||
# Extract videos from the page
|
||||
video_items = await self._extract_wix_videos(pw_page)
|
||||
media_items.extend(video_items)
|
||||
|
||||
# Also try to get images from network requests
|
||||
network_images = await self._extract_from_page_resources(pw_page)
|
||||
|
||||
# Merge and deduplicate
|
||||
for item in network_images:
|
||||
base_id = self._extract_wix_image_id(item.get('url', ''))
|
||||
if base_id and base_id not in self.seen_base_ids:
|
||||
self.seen_base_ids.add(base_id)
|
||||
media_items.append(item)
|
||||
|
||||
except Exception as e:
|
||||
print(f"WixHandler error: {e}")
|
||||
traceback.print_exc()
|
||||
|
||||
print(f"WixHandler: Extracted {len(media_items)} media items in {time.time() - start_time:.2f}s")
|
||||
return media_items
|
||||
|
||||
async def _scroll_page(self, pw_page):
|
||||
"""Scroll the page to trigger lazy loading"""
|
||||
try:
|
||||
# Get page height
|
||||
scroll_height = await pw_page.evaluate('document.body.scrollHeight')
|
||||
viewport_height = await pw_page.evaluate('window.innerHeight')
|
||||
|
||||
# Scroll in increments
|
||||
current_position = 0
|
||||
scroll_step = viewport_height * 0.8
|
||||
|
||||
while current_position < scroll_height:
|
||||
current_position += scroll_step
|
||||
await pw_page.evaluate(f'window.scrollTo(0, {current_position})')
|
||||
await pw_page.wait_for_timeout(500)
|
||||
|
||||
# Check if height increased (more content loaded)
|
||||
new_height = await pw_page.evaluate('document.body.scrollHeight')
|
||||
if new_height > scroll_height:
|
||||
scroll_height = new_height
|
||||
|
||||
# Scroll back to top
|
||||
await pw_page.evaluate('window.scrollTo(0, 0)')
|
||||
await pw_page.wait_for_timeout(500)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error scrolling page: {e}")
|
||||
|
||||
async def _extract_wix_videos(self, pw_page):
|
||||
"""
|
||||
Extract videos from Wix-powered pages.
|
||||
|
||||
Wix videos are hosted at video.wixstatic.com with this pattern:
|
||||
- https://video.wixstatic.com/video/{video_id}/1080p/mp4/file.mp4
|
||||
|
||||
Video IDs can be found from:
|
||||
1. Direct video src attributes
|
||||
2. Gallery items marked as video (poster images with 'f003' suffix)
|
||||
3. Data attributes in gallery containers
|
||||
"""
|
||||
media_items = []
|
||||
seen_video_ids = set()
|
||||
|
||||
try:
|
||||
# Extract video data from the page
|
||||
video_data = await pw_page.evaluate('''
|
||||
() => {
|
||||
const results = {
|
||||
directVideos: [],
|
||||
galleryVideos: [],
|
||||
videoContainers: []
|
||||
};
|
||||
|
||||
// 1. Get ALL video elements - check multiple src attributes
|
||||
const videos = document.querySelectorAll('video');
|
||||
for (const video of videos) {
|
||||
// Check multiple possible source attributes
|
||||
let src = video.src || video.currentSrc || '';
|
||||
const poster = video.poster || '';
|
||||
|
||||
// Also check source child elements
|
||||
if (!src) {
|
||||
const sourceEl = video.querySelector('source');
|
||||
if (sourceEl) src = sourceEl.src || '';
|
||||
}
|
||||
|
||||
// Any wixstatic video URL (including video.wixstatic.com)
|
||||
if (src && (src.includes('wixstatic.com') || src.includes('wix.com'))) {
|
||||
results.directVideos.push({
|
||||
src: src,
|
||||
poster: poster,
|
||||
type: 'direct',
|
||||
className: video.className
|
||||
});
|
||||
}
|
||||
// Poster images can help identify video IDs
|
||||
if (poster && poster.includes('wixstatic')) {
|
||||
results.directVideos.push({
|
||||
poster: poster,
|
||||
type: 'poster_reference'
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Get gallery items marked as videos
|
||||
const galleryVideoItems = document.querySelectorAll('[class*="gallery-item-video"]');
|
||||
for (const item of galleryVideoItems) {
|
||||
const img = item.querySelector('img');
|
||||
const video = item.querySelector('video');
|
||||
|
||||
// Get poster/thumbnail image which contains the video ID
|
||||
if (img && img.src && img.src.includes('wixstatic')) {
|
||||
// Video poster images typically end with f003 or f000
|
||||
const src = img.src;
|
||||
results.galleryVideos.push({
|
||||
posterSrc: src,
|
||||
hasVideoElement: !!video,
|
||||
type: 'gallery_video'
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Look for video containers with data attributes
|
||||
const videoContainers = document.querySelectorAll('[class*="video"]');
|
||||
for (const container of videoContainers) {
|
||||
const videoEl = container.querySelector('video[src*="video.wixstatic"]');
|
||||
if (videoEl) {
|
||||
results.videoContainers.push({
|
||||
src: videoEl.src,
|
||||
type: 'container'
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
''')
|
||||
|
||||
print(f"WixHandler: Found {len(video_data.get('directVideos', []))} direct videos, "
|
||||
f"{len(video_data.get('galleryVideos', []))} gallery videos, "
|
||||
f"{len(video_data.get('videoContainers', []))} container videos")
|
||||
|
||||
# Process direct videos (already have full URLs)
|
||||
for vid in video_data.get('directVideos', []):
|
||||
src = vid.get('src', '')
|
||||
print(f"WixHandler: Processing direct video - src: {src[:100] if src else 'None'}...")
|
||||
|
||||
# Check for any wixstatic video URL
|
||||
if src and ('video.wixstatic.com' in src or 'wixstatic.com' in src):
|
||||
video_id = self._extract_wix_video_id(src)
|
||||
print(f"WixHandler: Extracted video ID: {video_id}")
|
||||
|
||||
if video_id and video_id not in seen_video_ids:
|
||||
seen_video_ids.add(video_id)
|
||||
# Upgrade to 1080p if not already
|
||||
full_url = self._upgrade_wix_video_url(src)
|
||||
print(f"WixHandler: Upgraded video URL to: {full_url}")
|
||||
media_items.append({
|
||||
'url': full_url,
|
||||
'original_url': src,
|
||||
'type': 'video',
|
||||
'poster': vid.get('poster', ''),
|
||||
'source': 'wix_handler_video',
|
||||
'page_url': self.url, # For Referer header
|
||||
'qualities': {
|
||||
'1080p': f"https://video.wixstatic.com/video/{video_id}/1080p/mp4/file.mp4",
|
||||
'720p': f"https://video.wixstatic.com/video/{video_id}/720p/mp4/file.mp4",
|
||||
'480p': f"https://video.wixstatic.com/video/{video_id}/480p/mp4/file.mp4"
|
||||
}
|
||||
})
|
||||
|
||||
# Process gallery videos (construct URLs from poster images)
|
||||
for vid in video_data.get('galleryVideos', []):
|
||||
poster_src = vid.get('posterSrc', '')
|
||||
if poster_src:
|
||||
video_id = self._extract_video_id_from_poster(poster_src)
|
||||
if video_id and video_id not in seen_video_ids:
|
||||
seen_video_ids.add(video_id)
|
||||
# Construct direct MP4 URL
|
||||
video_url = f"https://video.wixstatic.com/video/{video_id}/1080p/mp4/file.mp4"
|
||||
media_items.append({
|
||||
'url': video_url,
|
||||
'type': 'video',
|
||||
'poster': poster_src,
|
||||
'source': 'wix_handler_gallery_video',
|
||||
'page_url': self.url, # For Referer header
|
||||
'qualities': {
|
||||
'1080p': f"https://video.wixstatic.com/video/{video_id}/1080p/mp4/file.mp4",
|
||||
'720p': f"https://video.wixstatic.com/video/{video_id}/720p/mp4/file.mp4",
|
||||
'480p': f"https://video.wixstatic.com/video/{video_id}/480p/mp4/file.mp4"
|
||||
}
|
||||
})
|
||||
print(f"WixHandler: Constructed video URL for ID {video_id}")
|
||||
|
||||
# Process video containers
|
||||
for vid in video_data.get('videoContainers', []):
|
||||
src = vid.get('src', '')
|
||||
if src and 'video.wixstatic.com' in src:
|
||||
video_id = self._extract_wix_video_id(src)
|
||||
if video_id and video_id not in seen_video_ids:
|
||||
seen_video_ids.add(video_id)
|
||||
full_url = self._upgrade_wix_video_url(src)
|
||||
media_items.append({
|
||||
'url': full_url,
|
||||
'original_url': src,
|
||||
'type': 'video',
|
||||
'source': 'wix_handler_container_video'
|
||||
})
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error extracting Wix videos: {e}")
|
||||
traceback.print_exc()
|
||||
|
||||
print(f"WixHandler: Extracted {len(media_items)} videos")
|
||||
return media_items
|
||||
|
||||
def _extract_wix_video_id(self, url):
|
||||
"""Extract video ID from a Wix video URL"""
|
||||
if not url:
|
||||
return None
|
||||
|
||||
# Pattern: video.wixstatic.com/video/{video_id}/quality/mp4/file.mp4
|
||||
# Example: video.wixstatic.com/video/b52b2f_19e70a48dd054abfb015bb2598c1ff8d/1080p/mp4/file.mp4
|
||||
match = re.search(r'video\.wixstatic\.com/video/([a-f0-9]+_[a-f0-9]+)', url)
|
||||
if match:
|
||||
return match.group(1)
|
||||
return None
|
||||
|
||||
def _extract_video_id_from_poster(self, poster_url):
|
||||
"""
|
||||
Extract video ID from a poster image URL.
|
||||
|
||||
Poster images have patterns like:
|
||||
- b52b2f_0a5d15df38dc4c08a18238b03b87082af003.jpg (f003 suffix = poster frame)
|
||||
- The video ID is the part before 'f003' or 'f000'
|
||||
"""
|
||||
if not poster_url:
|
||||
return None
|
||||
|
||||
# Pattern: media/{id}f003.jpg or media/{id}f000.jpg
|
||||
# We need to strip the f00X suffix to get the video ID
|
||||
patterns = [
|
||||
# Match ID with frame suffix (f003, f000, etc.)
|
||||
r'/media/([a-f0-9]+_[a-f0-9]+?)f\d{3}(?:~mv2)?\.(?:jpg|jpeg|png|webp)',
|
||||
r'/media/([a-f0-9]+_[a-f0-9]+?)f\d{3}\.(?:jpg|jpeg|png|webp)',
|
||||
# Direct ID pattern
|
||||
r'/media/([a-f0-9]+_[a-f0-9]{32})',
|
||||
]
|
||||
|
||||
for pattern in patterns:
|
||||
match = re.search(pattern, poster_url, re.IGNORECASE)
|
||||
if match:
|
||||
video_id = match.group(1)
|
||||
# Clean up any trailing frame indicator
|
||||
video_id = re.sub(r'f\d{3}$', '', video_id)
|
||||
return video_id
|
||||
|
||||
return None
|
||||
|
||||
def _upgrade_wix_video_url(self, url):
|
||||
"""Upgrade video URL to highest quality (1080p)"""
|
||||
if not url:
|
||||
return url
|
||||
|
||||
# Replace quality specifier with 1080p
|
||||
upgraded = re.sub(r'/\d{3,4}p/', '/1080p/', url)
|
||||
return upgraded
|
||||
|
||||
async def _extract_wix_images(self, pw_page):
|
||||
"""Extract and upgrade Wix images from the page"""
|
||||
media_items = []
|
||||
|
||||
try:
|
||||
# Get all images with their sources
|
||||
images = await pw_page.evaluate('''
|
||||
() => {
|
||||
const results = [];
|
||||
|
||||
// Get all img elements
|
||||
const imgs = document.querySelectorAll('img');
|
||||
for (const img of imgs) {
|
||||
const src = img.src || img.dataset.src || img.getAttribute('data-src') || '';
|
||||
const srcset = img.srcset || '';
|
||||
|
||||
if (src) {
|
||||
results.push({
|
||||
src: src,
|
||||
srcset: srcset,
|
||||
alt: img.alt || '',
|
||||
width: img.naturalWidth || img.width || 0,
|
||||
height: img.naturalHeight || img.height || 0
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Also check for background images in styles
|
||||
const allElements = document.querySelectorAll('*');
|
||||
for (const el of allElements) {
|
||||
const style = window.getComputedStyle(el);
|
||||
const bgImage = style.backgroundImage;
|
||||
if (bgImage && bgImage !== 'none' && bgImage.includes('wixstatic')) {
|
||||
const match = bgImage.match(/url\\(["']?([^"']+)["']?\\)/);
|
||||
if (match) {
|
||||
results.push({
|
||||
src: match[1],
|
||||
srcset: '',
|
||||
alt: '',
|
||||
width: 0,
|
||||
height: 0,
|
||||
isBackground: true
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check for Wix gallery data
|
||||
const galleryData = document.querySelectorAll('[data-image-info]');
|
||||
for (const el of galleryData) {
|
||||
try {
|
||||
const info = JSON.parse(el.getAttribute('data-image-info'));
|
||||
if (info && info.imageData && info.imageData.uri) {
|
||||
results.push({
|
||||
src: 'https://static.wixstatic.com/media/' + info.imageData.uri,
|
||||
srcset: '',
|
||||
alt: info.imageData.title || '',
|
||||
width: info.imageData.width || 0,
|
||||
height: info.imageData.height || 0,
|
||||
fromGalleryData: true
|
||||
});
|
||||
}
|
||||
} catch (e) {}
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
''')
|
||||
|
||||
print(f"WixHandler: Found {len(images)} image elements")
|
||||
|
||||
for img_data in images:
|
||||
src = img_data.get('src', '')
|
||||
srcset = img_data.get('srcset', '')
|
||||
|
||||
# Skip non-wix images and tracking pixels
|
||||
if not src:
|
||||
continue
|
||||
if 'wixstatic' not in src and 'wix.com' not in src:
|
||||
continue
|
||||
if img_data.get('width', 0) > 0 and img_data.get('width', 0) < 50:
|
||||
continue
|
||||
|
||||
# Extract base image ID
|
||||
base_id = self._extract_wix_image_id(src)
|
||||
if not base_id:
|
||||
continue
|
||||
|
||||
# Skip if we've already processed this image
|
||||
if base_id in self.seen_base_ids:
|
||||
continue
|
||||
self.seen_base_ids.add(base_id)
|
||||
|
||||
# Try to get highest resolution from srcset first
|
||||
best_url = self._get_best_from_srcset(srcset) if srcset else None
|
||||
|
||||
# Upgrade to maximum resolution
|
||||
if best_url:
|
||||
full_res_url = self._upgrade_wix_url(best_url)
|
||||
else:
|
||||
full_res_url = self._upgrade_wix_url(src)
|
||||
|
||||
if full_res_url in self.seen_urls:
|
||||
continue
|
||||
self.seen_urls.add(full_res_url)
|
||||
|
||||
media_item = {
|
||||
'url': full_res_url,
|
||||
'original_url': src,
|
||||
'type': 'image',
|
||||
'alt_text': img_data.get('alt', ''),
|
||||
'source': 'wix_handler',
|
||||
'original_width': img_data.get('width', 0),
|
||||
'original_height': img_data.get('height', 0)
|
||||
}
|
||||
media_items.append(media_item)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error extracting Wix images: {e}")
|
||||
traceback.print_exc()
|
||||
|
||||
return media_items
|
||||
|
||||
def _extract_wix_image_id(self, url):
|
||||
"""Extract the base image ID from a Wix URL"""
|
||||
if not url:
|
||||
return None
|
||||
|
||||
# Pattern: /media/[id]_[hash]~mv2.jpg or /media/[id].jpg
|
||||
# Example: b52b2f_1f9ee9ba5e6c4b4792257297be983580~mv2.jpg
|
||||
patterns = [
|
||||
r'/media/([a-f0-9]+_[a-f0-9]+[^/]*?)(?:/v1/|\.(?:jpg|jpeg|png|webp|gif))',
|
||||
r'/media/([a-f0-9]+_[^/]+?)(?:~mv2)?\.(?:jpg|jpeg|png|webp|gif)',
|
||||
r'/media/([a-f0-9_]+)(?:/|~|\.)',
|
||||
]
|
||||
|
||||
for pattern in patterns:
|
||||
match = re.search(pattern, url, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
return None
|
||||
|
||||
def _get_best_from_srcset(self, srcset):
|
||||
"""Parse srcset and return the highest resolution URL"""
|
||||
if not srcset:
|
||||
return None
|
||||
|
||||
best_url = None
|
||||
best_width = 0
|
||||
|
||||
# Parse srcset: "url1 800w, url2 1200w, ..."
|
||||
parts = srcset.split(',')
|
||||
for part in parts:
|
||||
part = part.strip()
|
||||
match = re.match(r'(\S+)\s+(\d+)w', part)
|
||||
if match:
|
||||
url, width = match.groups()
|
||||
width = int(width)
|
||||
if width > best_width:
|
||||
best_width = width
|
||||
best_url = url
|
||||
|
||||
return best_url
|
||||
|
||||
def _upgrade_wix_url(self, url):
|
||||
"""
|
||||
Upgrade a Wix image URL to maximum resolution.
|
||||
|
||||
Strategies:
|
||||
1. Request original by removing transforms
|
||||
2. Request very large dimensions
|
||||
3. Change quality to maximum
|
||||
"""
|
||||
if not url:
|
||||
return url
|
||||
|
||||
# If it's already an original URL (no /v1/fill/ or /v1/fit/), return as-is
|
||||
if '/v1/fill/' not in url and '/v1/crop/' not in url and '/v1/fit/' not in url:
|
||||
# Try to ensure we get the raw image
|
||||
return url
|
||||
|
||||
try:
|
||||
# Extract the base media URL and transform it
|
||||
# Pattern: https://static.wixstatic.com/media/[id]/v1/fill/w_X,h_Y,.../filename.ext
|
||||
# Or: https://static.wixstatic.com/media/[id]/v1/fit/w_X,h_Y,.../filename.ext
|
||||
|
||||
# Strategy 1: Request maximum resolution (4K+)
|
||||
# Replace dimension parameters with large values
|
||||
upgraded = url
|
||||
|
||||
# Replace width with large value (Wix typically supports up to 5000)
|
||||
upgraded = re.sub(r'w_\d+', 'w_4000', upgraded)
|
||||
|
||||
# Replace height with large value
|
||||
upgraded = re.sub(r'h_\d+', 'h_5000', upgraded)
|
||||
|
||||
# Set quality to maximum (90 is usually best balance, 100 can cause issues)
|
||||
upgraded = re.sub(r'q_\d+', 'q_95', upgraded)
|
||||
|
||||
# Change fit to fill for better quality (fit may letterbox)
|
||||
# Actually keep as-is since fit respects aspect ratio
|
||||
|
||||
# Prefer JPEG over AVIF/WebP for better compatibility and quality
|
||||
upgraded = re.sub(r'enc_avif', 'enc_auto', upgraded)
|
||||
upgraded = re.sub(r'enc_webp', 'enc_auto', upgraded)
|
||||
|
||||
# Remove quality_auto which adds compression
|
||||
upgraded = re.sub(r',quality_auto', '', upgraded)
|
||||
|
||||
print(f"WixHandler: Upgraded URL dimensions to 4000x5000")
|
||||
|
||||
return upgraded
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error upgrading Wix URL: {e}")
|
||||
return url
|
||||
|
||||
async def _extract_from_page_resources(self, pw_page):
|
||||
"""Extract images from page resources/network requests"""
|
||||
media_items = []
|
||||
|
||||
try:
|
||||
# Get all loaded resources
|
||||
resources = await pw_page.evaluate('''
|
||||
() => {
|
||||
const resources = [];
|
||||
|
||||
// Check performance entries for loaded resources
|
||||
if (window.performance && window.performance.getEntriesByType) {
|
||||
const entries = window.performance.getEntriesByType('resource');
|
||||
for (const entry of entries) {
|
||||
if (entry.initiatorType === 'img' ||
|
||||
entry.name.match(/\\.(jpg|jpeg|png|webp|gif|avif)/i)) {
|
||||
resources.push(entry.name);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return resources;
|
||||
}
|
||||
''')
|
||||
|
||||
for url in resources:
|
||||
if 'wixstatic' not in url:
|
||||
continue
|
||||
|
||||
base_id = self._extract_wix_image_id(url)
|
||||
if not base_id or base_id in self.seen_base_ids:
|
||||
continue
|
||||
|
||||
full_res_url = self._upgrade_wix_url(url)
|
||||
|
||||
if full_res_url not in self.seen_urls:
|
||||
self.seen_urls.add(full_res_url)
|
||||
media_items.append({
|
||||
'url': full_res_url,
|
||||
'original_url': url,
|
||||
'type': 'image',
|
||||
'source': 'wix_handler_network'
|
||||
})
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error extracting from page resources: {e}")
|
||||
|
||||
return media_items
|
||||
|
||||
async def _get_playwright_page_async(self, page):
|
||||
"""Get the underlying Playwright page object"""
|
||||
if not PLAYWRIGHT_AVAILABLE:
|
||||
return None
|
||||
|
||||
# Handle different page wrapper types
|
||||
if hasattr(page, '_playwright_page'):
|
||||
return page._playwright_page
|
||||
elif hasattr(page, 'page'):
|
||||
return page.page
|
||||
elif isinstance(page, AsyncPage) if AsyncPage else False:
|
||||
return page
|
||||
elif hasattr(page, 'playwright_page'):
|
||||
return page.playwright_page
|
||||
|
||||
return page
|
||||
|
||||
async def extract_with_direct_playwright(self, page, **kwargs):
|
||||
"""Direct Playwright extraction method called by the scraper"""
|
||||
return await self.extract_media_items_async(page)
|
||||
@@ -0,0 +1,19 @@
|
||||
"""
|
||||
Utils Package
|
||||
|
||||
Utility modules for download_tools nodes.
|
||||
"""
|
||||
|
||||
from .persistent_settings import (
|
||||
PersistentSettings,
|
||||
get_settings_manager,
|
||||
get_persistent_setting,
|
||||
set_persistent_setting
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
'PersistentSettings',
|
||||
'get_settings_manager',
|
||||
'get_persistent_setting',
|
||||
'set_persistent_setting'
|
||||
]
|
||||
@@ -0,0 +1,167 @@
|
||||
"""
|
||||
Persistent Settings Manager
|
||||
|
||||
Description: Manages persistent settings for download_tools nodes across reloads
|
||||
Author: Eric Hiss (GitHub: EricRollei)
|
||||
Contact: eric@historic.camera, eric@rollei.us
|
||||
License: Dual License (Non-Commercial and Commercial Use)
|
||||
Copyright (c) 2025 Eric Hiss. All rights reserved.
|
||||
"""
|
||||
|
||||
import os
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Optional, Any, Dict
|
||||
|
||||
|
||||
class PersistentSettings:
|
||||
"""
|
||||
Manages persistent settings for download_tools nodes.
|
||||
Settings are stored in a JSON file and persist across node/workflow reloads.
|
||||
"""
|
||||
|
||||
_instance = None
|
||||
_settings: Dict[str, Any] = {}
|
||||
_settings_file: Path = None
|
||||
|
||||
def __new__(cls):
|
||||
"""Singleton pattern to ensure only one instance exists."""
|
||||
if cls._instance is None:
|
||||
cls._instance = super().__new__(cls)
|
||||
cls._instance._initialized = False
|
||||
return cls._instance
|
||||
|
||||
def __init__(self):
|
||||
"""Initialize the settings manager."""
|
||||
if self._initialized:
|
||||
return
|
||||
|
||||
# Determine settings file path
|
||||
self._settings_file = Path(__file__).parent.parent / "configs" / "node_settings.json"
|
||||
self._settings = self._load_settings()
|
||||
self._initialized = True
|
||||
print(f"PersistentSettings initialized from: {self._settings_file}")
|
||||
|
||||
def _load_settings(self) -> Dict[str, Any]:
|
||||
"""Load settings from the JSON file."""
|
||||
default_settings = {
|
||||
"web_scraper": {
|
||||
"auth_config_path": ""
|
||||
},
|
||||
"gallery_dl": {
|
||||
"config_path": "",
|
||||
"cookie_file": ""
|
||||
},
|
||||
"yt_dlp": {
|
||||
"config_path": ""
|
||||
}
|
||||
}
|
||||
|
||||
try:
|
||||
if self._settings_file.exists():
|
||||
with open(self._settings_file, 'r', encoding='utf-8') as f:
|
||||
loaded = json.load(f)
|
||||
# Merge with defaults to ensure all keys exist
|
||||
for key in default_settings:
|
||||
if key not in loaded:
|
||||
loaded[key] = default_settings[key]
|
||||
elif isinstance(default_settings[key], dict):
|
||||
for subkey in default_settings[key]:
|
||||
if subkey not in loaded[key]:
|
||||
loaded[key][subkey] = default_settings[key][subkey]
|
||||
return loaded
|
||||
else:
|
||||
# Create the file with defaults
|
||||
self._settings_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(self._settings_file, 'w', encoding='utf-8') as f:
|
||||
json.dump(default_settings, f, indent=4)
|
||||
return default_settings
|
||||
except Exception as e:
|
||||
print(f"Error loading persistent settings: {e}")
|
||||
return default_settings
|
||||
|
||||
def _save_settings(self) -> bool:
|
||||
"""Save current settings to the JSON file."""
|
||||
try:
|
||||
self._settings_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(self._settings_file, 'w', encoding='utf-8') as f:
|
||||
json.dump(self._settings, f, indent=4)
|
||||
return True
|
||||
except Exception as e:
|
||||
print(f"Error saving persistent settings: {e}")
|
||||
return False
|
||||
|
||||
def get(self, node_type: str, key: str, default: Any = "") -> Any:
|
||||
"""
|
||||
Get a setting value for a specific node type.
|
||||
|
||||
Args:
|
||||
node_type: The type of node ('web_scraper', 'gallery_dl', 'yt_dlp')
|
||||
key: The setting key to retrieve
|
||||
default: Default value if setting not found
|
||||
|
||||
Returns:
|
||||
The setting value or default
|
||||
"""
|
||||
try:
|
||||
if node_type in self._settings and key in self._settings[node_type]:
|
||||
value = self._settings[node_type][key]
|
||||
return value if value else default
|
||||
return default
|
||||
except Exception:
|
||||
return default
|
||||
|
||||
def set(self, node_type: str, key: str, value: Any) -> bool:
|
||||
"""
|
||||
Set a setting value for a specific node type.
|
||||
|
||||
Args:
|
||||
node_type: The type of node ('web_scraper', 'gallery_dl', 'yt_dlp')
|
||||
key: The setting key to set
|
||||
value: The value to store
|
||||
|
||||
Returns:
|
||||
True if successful, False otherwise
|
||||
"""
|
||||
try:
|
||||
if node_type not in self._settings:
|
||||
self._settings[node_type] = {}
|
||||
|
||||
# Only update if value is non-empty (don't overwrite with empty values)
|
||||
if value and str(value).strip():
|
||||
self._settings[node_type][key] = str(value).strip()
|
||||
return self._save_settings()
|
||||
return True
|
||||
except Exception as e:
|
||||
print(f"Error setting persistent setting: {e}")
|
||||
return False
|
||||
|
||||
def get_all(self, node_type: str) -> Dict[str, Any]:
|
||||
"""Get all settings for a specific node type."""
|
||||
return self._settings.get(node_type, {})
|
||||
|
||||
def reload(self) -> None:
|
||||
"""Reload settings from file (useful if file was edited externally)."""
|
||||
self._settings = self._load_settings()
|
||||
|
||||
|
||||
# Global instance for easy access
|
||||
_settings_manager: Optional[PersistentSettings] = None
|
||||
|
||||
|
||||
def get_settings_manager() -> PersistentSettings:
|
||||
"""Get the global settings manager instance."""
|
||||
global _settings_manager
|
||||
if _settings_manager is None:
|
||||
_settings_manager = PersistentSettings()
|
||||
return _settings_manager
|
||||
|
||||
|
||||
def get_persistent_setting(node_type: str, key: str, default: Any = "") -> Any:
|
||||
"""Convenience function to get a persistent setting."""
|
||||
return get_settings_manager().get(node_type, key, default)
|
||||
|
||||
|
||||
def set_persistent_setting(node_type: str, key: str, value: Any) -> bool:
|
||||
"""Convenience function to set a persistent setting."""
|
||||
return get_settings_manager().set(node_type, key, value)
|
||||
Reference in New Issue
Block a user