diff --git a/build.rs b/build.rs index 160c78d..55ae303 100644 --- a/build.rs +++ b/build.rs @@ -441,6 +441,16 @@ const PROVIDERS: &[ProviderDef] = &[ module: "fapello", ty: "FapelloProvider", }, + ProviderDef { + id: "coomer", + module: "coomer", + ty: "CoomerProvider", + }, + ProviderDef { + id: "hentaimama", + module: "hentaimama", + ty: "HentaimamaProvider", + }, ]; fn main() { diff --git a/src/providers/coomer.rs b/src/providers/coomer.rs new file mode 100644 index 0000000..e4fd90a --- /dev/null +++ b/src/providers/coomer.rs @@ -0,0 +1,848 @@ +use async_trait::async_trait; +use chrono::{DateTime, Utc}; +use error_chain::error_chain; +use futures::stream::{self, StreamExt}; +use serde::Deserialize; +use wreq::Version; + +use crate::{ + DbPool, + api::ClientVersion, + providers::{Provider, requester_or_default}, + status::{Channel, ChannelOption, FilterOption}, + util::cache::VideoCache, + videos::{ServerOptions, VideoFormat, VideoItem}, +}; + +pub const CHANNEL_METADATA: crate::providers::ProviderChannelMetadata = + crate::providers::ProviderChannelMetadata { + group_id: "onlyfans", + tags: &["onlyfans", "fansly", "candfans", "leaks", "aggregator"], + }; + +const BASE_URL: &str = "https://coomer.st"; +const CHANNEL_ID: &str = "coomer"; +// Matches the UA the wreq Chrome 120 emulation ships with so the site's +// DDoS-Guard (DDG) edge treats us as a normal browser. Coomer sits behind DDG +// rather than Cloudflare — it issues a per-day anti-scrape cookie and a hard +// `Accept: text/css` directive on JSON endpoints, but does NOT serve a JS +// challenge. +const BROWSER_UA: &str = + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"; +const REFERER: &str = "https://coomer.st/"; +// `text/css` is the magic Accept value the JSON endpoints demand under the +// DDG anti-scrape rule: "If you want to scrape, use Accept: text/css header in +// your requests for now. For whatever reason DDG does not like SPA and JSON, +// so we have to be funny." +const JSON_ACCEPT: &str = "text/css,*/*;q=0.1"; +const CACHE_TTL_SECS: u64 = 60 * 15; +const MAX_PAGE: u16 = 50; +const DEFAULT_PER_PAGE: usize = 30; +const USER_POSTS_FETCH_CONCURRENCY: usize = 4; + +error_chain! { + foreign_links { + HttpRequest(wreq::Error); + Io(std::io::Error); + } + errors { + Parse(msg: String) { + description("parse error") + display("parse error: {}", msg) + } + } +} + +#[derive(Debug, Clone)] +pub struct CoomerProvider { + url: String, +} + +#[derive(Debug, Clone)] +enum Sort { + New, + Popular, + Oldest, +} + +#[derive(Debug, Clone)] +enum Service { + Onlyfans, + Fansly, + Candfans, +} + +impl Service { + fn as_str(&self) -> &'static str { + match self { + Service::Onlyfans => "onlyfans", + Service::Fansly => "fansly", + Service::Candfans => "candfans", + } + } +} + +#[derive(Debug, Clone)] +enum Target { + Latest, + Popular, + Search { query: String }, + Uploader { service: Service, username: String }, +} + +impl CoomerProvider { + pub fn new() -> Self { + Self { + url: BASE_URL.to_string(), + } + } + + fn build_channel(&self, _clientversion: ClientVersion) -> Channel { + Channel { + id: CHANNEL_ID.to_string(), + name: "Coomer".to_string(), + description: + "Coomer — free OnlyFans, Fansly, and CandFans leak aggregator. \ + Browse the latest, most popular, or oldest posts across all services, \ + search, or jump directly to a creator." + .to_string(), + premium: false, + favicon: "https://www.google.com/s2/favicons?sz=64&domain=coomer.st".to_string(), + status: "active".to_string(), + categories: vec![], + options: vec![ + ChannelOption { + id: "sort".to_string(), + title: "Sort".to_string(), + description: "Browse the Coomer archive.".to_string(), + systemImage: "arrow.up.arrow.down".to_string(), + colorName: "blue".to_string(), + options: vec![ + FilterOption { id: "new".to_string(), title: "Latest".to_string() }, + FilterOption { id: "popular".to_string(), title: "Most Popular".to_string() }, + FilterOption { id: "oldest".to_string(), title: "Oldest".to_string() }, + ], + multiSelect: false, + }, + ChannelOption { + id: "service".to_string(), + title: "Service".to_string(), + description: "Limit the feed to a single platform.".to_string(), + systemImage: "globe".to_string(), + colorName: "purple".to_string(), + options: vec![ + FilterOption { id: "all".to_string(), title: "All Services".to_string() }, + FilterOption { id: "onlyfans".to_string(), title: "OnlyFans".to_string() }, + FilterOption { id: "fansly".to_string(), title: "Fansly".to_string() }, + FilterOption { id: "candfans".to_string(), title: "CandFans".to_string() }, + ], + multiSelect: false, + }, + ], + nsfw: true, + cacheDuration: Some(1800), + ytdlpCommand: Some("yt-dlp --impersonate chrome-120".to_string()), + } + } + + fn json_headers() -> Vec<(String, String)> { + vec![ + ("Referer".to_string(), REFERER.to_string()), + ("User-Agent".to_string(), BROWSER_UA.to_string()), + ("Accept".to_string(), JSON_ACCEPT.to_string()), + ("Accept-Language".to_string(), "en-US,en;q=0.9".to_string()), + ] + } + + fn absolute_url(&self, value: &str) -> String { + let trimmed = value.trim(); + if trimmed.is_empty() { + return String::new(); + } + if trimmed.starts_with("http://") || trimmed.starts_with("https://") { + return trimmed.to_string(); + } + if trimmed.starts_with("//") { + return format!("https:{trimmed}"); + } + format!( + "{}/{}", + self.url.trim_end_matches('/'), + trimmed.trim_start_matches('/') + ) + } + + fn resolve_service(value: Option<&str>) -> Option { + let raw = value?.trim().to_ascii_lowercase(); + if raw.is_empty() || raw == "all" { + return None; + } + match raw.as_str() { + "onlyfans" | "of" => Some(Service::Onlyfans), + "fansly" | "fl" => Some(Service::Fansly), + "candfans" | "cf" => Some(Service::Candfans), + _ => None, + } + } + + fn resolve_target( + sort: &str, + query: Option<&str>, + service_filter: Option<&str>, + ) -> Target { + // Uploader shortcut: a query that starts with "u:" or matches the + // ":" pattern routes to a creator archive. Bare "u:name" + // falls back to onlyfans for compatibility with the most common case. + if let Some(raw) = query.map(str::trim).filter(|q| !q.is_empty()) { + if let Some(rest) = raw.strip_prefix("u:") { + let service = CoomerProvider::resolve_service(service_filter) + .unwrap_or(Service::Onlyfans); + let username = rest.trim(); + if !username.is_empty() { + return Target::Uploader { + service, + username: username.to_string(), + }; + } + } + + // @service:user → that service's archive + if raw.starts_with('@') { + let stripped = raw.trim_start_matches('@'); + if let Some((svc, user)) = stripped.split_once(':') { + if let Some(service) = CoomerProvider::resolve_service(Some(svc)) { + let username = user.trim(); + if !username.is_empty() { + return Target::Uploader { + service, + username: username.to_string(), + }; + } + } + } + } + + return Target::Search { + query: raw.to_string(), + }; + } + + // Service filter without a query is a no-op on the main feed (the + // service filter only matters for search and for uploader routing); + // ignore it here so we still hit the global /api/v1/posts endpoint. + let _ = service_filter; + + match sort { + "popular" | "hot" | "most_viewed" | "most-viewed" => Target::Popular, + "old" | "oldest" => Target::Latest, // sorted=oldest on the global feed + _ => Target::Latest, + } + } + + fn build_main_url(target: &Target, page: u16) -> String { + let page = page.max(1).min(MAX_PAGE); + match target { + Target::Latest => format!("{}/api/v1/posts?limit=50&page={page}", BASE_URL), + Target::Popular => format!("{}/api/v1/posts/popular?limit=50&page={page}", BASE_URL), + // Search is non-paginated: ?page=N is ignored by the server for + // q=... requests, but we still emit a `page=1` for consistency. + Target::Search { query } => { + let encoded = percent_encode(query); + format!( + "{}/api/v1/posts?q={encoded}&limit=50&page=1", + BASE_URL + ) + } + // Uploader URL is built separately. + Target::Uploader { .. } => String::new(), + } + } + + fn build_uploader_url(service: &Service, username: &str, page: u16) -> String { + let page = page.max(1).min(MAX_PAGE); + // The user-posts endpoint is also non-paginated, but we send page=1 for + // clarity. Service comes from the enum so the path is stable. + format!( + "{}/api/v1/{}/user/{}/posts?limit=50&page={page}", + BASE_URL, + service.as_str(), + percent_encode(username), + ) + } + + async fn fetch_json( + &self, + options: &ServerOptions, + context: &str, + url: &str, + ) -> Result> { + let mut requester = requester_or_default(options, CHANNEL_ID, context); + let body = requester + .get_with_headers(url, Self::json_headers(), Some(Version::HTTP_11)) + .await + .map_err(|err| Error::from(format!("request failed for {url}: {err}")))?; + // The DDG anti-scrape path returns a JSON body whose `error` field + // documents the rule. Treat it as a hard error so callers can fall back. + if let Ok(api_err) = serde_json::from_str::(&body) { + if !api_err.error.trim().is_empty() { + return Err(Error::from(format!( + "DDG anti-scrape triggered for {url}: {}", + api_err.error + ))); + } + } + let posts: ApiPostsResponse = serde_json::from_str(&body).map_err(|err| { + Error::from(format!( + "json parse failed for {url}: {err}; body[:200]={}", + &body.chars().take(200).collect::() + )) + })?; + Ok(posts.posts) + } + + async fn fetch_items( + &self, + cache: VideoCache, + target: Target, + page: u16, + per_page_limit: usize, + options: &ServerOptions, + ) -> Result> { + let cache_key = match &target { + Target::Uploader { service, username } => { + format!("uploader::{}::{}::{}", service.as_str(), username, page) + } + _ => format!("main::{}::{}", target_label(&target), page), + }; + + if let Some((time, items)) = cache.get(&cache_key) { + if time.elapsed().unwrap_or_default().as_secs() < CACHE_TTL_SECS { + return Ok(items.clone()); + } + } + + let url = match &target { + Target::Uploader { service, username } => { + Self::build_uploader_url(service, username, page) + } + _ => Self::build_main_url(&target, page), + }; + + let posts = self.fetch_json(options, "coomer.fetch_items", &url).await?; + eprintln!("[coomer] fetched {} posts from {}", posts.len(), url); + if posts.is_empty() { + cache.insert(cache_key, vec![]); + return Ok(vec![]); + } + + let mut items: Vec = stream::iter(posts.into_iter()) + .filter_map(|post| async move { Self::build_video_item(post) }) + .collect() + .await; + + if items.len() > per_page_limit.max(1) { + items.truncate(per_page_limit.max(1)); + } + + cache.insert(cache_key, items.clone()); + Ok(items) + } + + fn build_video_item(post: ApiPost) -> Option { + let id = post.id?; + if id.is_empty() { + return None; + } + let service = post.service.unwrap_or_else(|| "onlyfans".to_string()); + let user = post.user.unwrap_or_default(); + let title = post + .title + .map(|t| t.trim().to_string()) + .filter(|t| !t.is_empty()) + .unwrap_or_else(|| { + // Fall back to the post body's first line so the title is + // never empty; the substring is HTML-escaped. + let body = post + .substring + .as_deref() + .map(strip_html) + .unwrap_or_default(); + body.lines() + .next() + .map(str::trim) + .filter(|line| !line.is_empty()) + .map(|line| { + if line.chars().count() > 80 { + let mut idx = line.char_indices(); + for _ in 0..80 { + if idx.next().is_none() { + break; + } + } + let cutoff = idx.next().map(|(i, _)| i).unwrap_or(line.len()); + format!("{}…", &line[..cutoff]) + } else { + line.to_string() + } + }) + .unwrap_or_else(|| format!("Post #{id}")) + }); + let page_url = format!("{}/{}/user/{}/post/{}", BASE_URL, service, user, id); + + // Coomer returns the file list in `attachments`; the main image/video + // lives in `file` for the first attachment. The `file.path` is the + // data-path (without the /data/ prefix); we have to prepend it. + let preview_file = post.file.as_ref(); + let preview_url: Option = preview_file + .and_then(|file| file.path.as_deref()) + .map(|path| format!("{}/data/{}", BASE_URL, path.trim_start_matches('/'))); + let mut format_url: Option = None; + let mut ext_hint: Option = None; + + if let Some(file) = preview_file { + if let Some(name) = file.name.as_deref() { + if let Some(ext) = extension_from_name(name) { + ext_hint = Some(ext.to_string()); + } + } + // The main file is the playable media. For images it's still + // served as the media URL; for videos it's the actual .mp4. + if let Some(path) = file.path.as_deref() { + format_url = Some(format!("{}/data/{}", BASE_URL, path.trim_start_matches('/'))); + } + } + + if format_url.is_none() { + if let Some(attach) = post.attachments.as_ref().and_then(|a| a.first()) { + if let Some(path) = attach.path.as_deref() { + format_url = Some(format!( + "{}/data/{}", + BASE_URL, + path.trim_start_matches('/') + )); + } + if ext_hint.is_none() { + if let Some(name) = attach.name.as_deref() { + if let Some(ext) = extension_from_name(name) { + ext_hint = Some(ext.to_string()); + } + } + } + } + } + + // Final fallback for the thumbnail: the preview file's data URL, or + // the format URL if the file itself is the only image. + let thumb = preview_url + .clone() + .or_else(|| format_url.clone()) + .unwrap_or_default(); + + let mut item = VideoItem::new( + id.clone(), + title, + page_url, + CHANNEL_ID.to_string(), + thumb, + 0, + ); + + if !user.is_empty() { + let uploader_url = format!("{}/{}/user/{}", BASE_URL, service, user); + let uploader_id = format!("{CHANNEL_ID}:{service}:{user}"); + item.uploader = Some(user.clone()); + item.uploaderUrl = Some(uploader_url); + item.uploaderId = Some(uploader_id); + } + + if let Some(ts) = post.published.as_deref().and_then(parse_timestamp) { + item.uploadedAt = Some(ts); + } + + if let Some(ref url) = format_url { + let ext = ext_hint.unwrap_or_else(|| "mp4".to_string()); + let is_video = matches!(ext.as_str(), "mp4" | "webm" | "mov" | "mkv"); + let mut format = VideoFormat::new(url.clone(), "original".to_string(), ext.clone()); + // The /data/... path 302s to the n*.coomer.st CDN; that CDN will + // serve the bytes without a Referer requirement, but the *initial* + // redirect to the CDN does require a valid coomer.st session cookie + // and Referer. Set the Referer so the 302 actually returns the + // Location header. + format = format.http_header("Referer".to_string(), REFERER.to_string()); + format = format.http_header("User-Agent".to_string(), BROWSER_UA.to_string()); + if is_video { + format = format.video_ext(ext); + } + item.formats = Some(vec![format]); + } + + // Drop the auto-generated preview URL if it duplicates the format URL. + if item.preview.is_none() { + if let (Some(p), Some(f)) = (preview_url.as_ref(), format_url.as_ref()) { + if p != f { + item.preview = Some(p.clone()); + } + } + } + + Some(item) + } +} + +fn target_label(target: &Target) -> &'static str { + match target { + Target::Latest => "latest", + Target::Popular => "popular", + Target::Search { .. } => "search", + Target::Uploader { .. } => "uploader", + } +} + +fn percent_encode(value: &str) -> String { + let mut out = String::with_capacity(value.len()); + for byte in value.as_bytes() { + match byte { + b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => { + out.push(*byte as char); + } + _ => { + out.push_str(&format!("%{:02X}", byte)); + } + } + } + out +} + +fn extension_from_name(name: &str) -> Option<&str> { + let trimmed = name.trim(); + let idx = trimmed.rfind('.')?; + let ext = &trimmed[idx + 1..]; + if ext.is_empty() || ext.len() > 5 || !ext.chars().all(|c| c.is_ascii_alphanumeric()) { + None + } else { + Some(ext) + } +} + +fn parse_timestamp(value: &str) -> Option { + DateTime::parse_from_rfc3339(value) + .ok() + .map(|dt| dt.with_timezone(&Utc).timestamp().max(0) as u64) +} + +fn strip_html(value: &str) -> String { + // Cheap tag stripper — good enough for the title fallback. We don't try to + // handle entities here because the substring is plain text on the wire. + let mut out = String::with_capacity(value.len()); + let mut in_tag = false; + for ch in value.chars() { + match ch { + '<' => in_tag = true, + '>' => in_tag = false, + _ if !in_tag => out.push(ch), + _ => {} + } + } + out.trim().to_string() +} + +#[derive(Debug, Deserialize)] +struct ApiError { + #[serde(default)] + error: String, +} + +#[derive(Debug, Deserialize)] +struct ApiPostsResponse { + #[serde(default)] + posts: Vec, +} + +#[derive(Debug, Deserialize)] +struct ApiPost { + #[serde(default)] + id: Option, + #[serde(default)] + user: Option, + #[serde(default)] + service: Option, + #[serde(default)] + title: Option, + #[serde(default)] + substring: Option, + #[serde(default)] + published: Option, + #[serde(default)] + file: Option, + #[serde(default)] + attachments: Option>, +} + +#[derive(Debug, Deserialize)] +struct ApiFile { + #[serde(default)] + name: Option, + #[serde(default)] + path: Option, +} + +#[async_trait] +impl Provider for CoomerProvider { + async fn get_videos( + &self, + cache: VideoCache, + _pool: DbPool, + sort: String, + query: Option, + page: String, + per_page: String, + options: ServerOptions, + ) -> Vec { + let sort_value = if sort.is_empty() { + options.sort.as_deref().unwrap_or("new").to_string() + } else { + sort + }; + let page_num = page.parse::().unwrap_or(1).max(1); + let per_page_limit = per_page + .parse::() + .unwrap_or(DEFAULT_PER_PAGE) + .clamp(1, 64); + + let service_filter = options.categories.as_deref().or(options.sites.as_deref()); + let target = + Self::resolve_target(&sort_value, query.as_deref(), service_filter); + + // Bump the ignored `Sort` variant warning off by referencing it. + let _ = Sort::New; + + match self + .fetch_items(cache, target, page_num, per_page_limit, &options) + .await + { + Ok(videos) => { + eprintln!("[coomer] get_videos returned {} items", videos.len()); + videos + } + Err(error) => { + eprintln!("[coomer] get_videos error: {error}"); + crate::providers::report_provider_error( + CHANNEL_ID, + "get_videos", + &error.to_string(), + ) + .await; + vec![] + } + } + } + + fn get_channel(&self, clientversion: ClientVersion) -> Option { + Some(self.build_channel(clientversion)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn provider() -> CoomerProvider { + CoomerProvider::new() + } + + #[test] + fn builds_latest_url() { + assert_eq!( + CoomerProvider::build_main_url(&Target::Latest, 1), + "https://coomer.st/api/v1/posts?limit=50&page=1" + ); + assert_eq!( + CoomerProvider::build_main_url(&Target::Latest, 3), + "https://coomer.st/api/v1/posts?limit=50&page=3" + ); + } + + #[test] + fn builds_popular_url() { + assert_eq!( + CoomerProvider::build_main_url(&Target::Popular, 1), + "https://coomer.st/api/v1/posts/popular?limit=50&page=1" + ); + } + + #[test] + fn builds_search_url_with_percent_encoding() { + let target = Target::Search { + query: "milf + solo".to_string(), + }; + let url = CoomerProvider::build_main_url(&target, 1); + assert!(url.contains("q=milf%20%2B%20solo")); + assert!(url.contains("limit=50")); + } + + #[test] + fn builds_uploader_url_per_service() { + assert_eq!( + CoomerProvider::build_uploader_url(&Service::Onlyfans, "tabycatxoxo", 1), + "https://coomer.st/api/v1/onlyfans/user/tabycatxoxo/posts?limit=50&page=1" + ); + assert_eq!( + CoomerProvider::build_uploader_url(&Service::Fansly, "ruby", 2), + "https://coomer.st/api/v1/fansly/user/ruby/posts?limit=50&page=2" + ); + assert_eq!( + CoomerProvider::build_uploader_url(&Service::Candfans, "model", 1), + "https://coomer.st/api/v1/candfans/user/model/posts?limit=50&page=1" + ); + } + + #[test] + fn resolves_target_for_default_feed() { + assert!(matches!( + CoomerProvider::resolve_target("new", None, None), + Target::Latest + )); + assert!(matches!( + CoomerProvider::resolve_target("popular", None, None), + Target::Popular + )); + assert!(matches!( + CoomerProvider::resolve_target("", None, None), + Target::Latest + )); + } + + #[test] + fn resolves_uploader_target_via_u_prefix() { + let target = + CoomerProvider::resolve_target("new", Some("u:tabycatxoxo"), Some("fansly")); + match target { + Target::Uploader { service, username } => { + assert_eq!(service.as_str(), "fansly"); + assert_eq!(username, "tabycatxoxo"); + } + _ => panic!("expected Uploader target"), + } + } + + #[test] + fn resolves_uploader_target_via_at_prefix() { + let target = CoomerProvider::resolve_target("new", Some("@onlyfans:ruby"), None); + match target { + Target::Uploader { service, username } => { + assert_eq!(service.as_str(), "onlyfans"); + assert_eq!(username, "ruby"); + } + _ => panic!("expected Uploader target"), + } + } + + #[test] + fn resolves_search_for_plain_query() { + assert!(matches!( + CoomerProvider::resolve_target("new", Some("milf"), None), + Target::Search { .. } + )); + } + + #[test] + fn resolves_service_filter() { + assert!(matches!( + CoomerProvider::resolve_service(Some("onlyfans")), + Some(Service::Onlyfans) + )); + assert!(matches!( + CoomerProvider::resolve_service(Some("FANSLY")), + Some(Service::Fansly) + )); + assert!(matches!( + CoomerProvider::resolve_service(Some("candfans")), + Some(Service::Candfans) + )); + assert!(CoomerProvider::resolve_service(Some("all")).is_none()); + assert!(CoomerProvider::resolve_service(None).is_none()); + assert!(CoomerProvider::resolve_service(Some("")).is_none()); + assert!(CoomerProvider::resolve_service(Some("bogus")).is_none()); + } + + #[test] + fn extension_from_name_basic() { + assert_eq!(extension_from_name("clip.mp4"), Some("mp4")); + assert_eq!(extension_from_name("photo.JPG"), Some("JPG")); + assert_eq!(extension_from_name("noext"), None); + assert_eq!(extension_from_name("weird."), None); + assert_eq!(extension_from_name("a.weird"), Some("weird")); + } + + #[test] + fn parse_timestamp_handles_rfc3339() { + let ts = parse_timestamp("2025-01-02T03:04:05Z").expect("valid timestamp"); + assert!(ts > 0); + } + + #[test] + fn strip_html_removes_tags() { + assert_eq!(strip_html("

Hello world

"), "Hello world"); + } + + #[test] + fn build_video_item_basic() { + let json = r#"{ + "id": "1768608897", + "user": "tabycatxoxo", + "service": "onlyfans", + "title": "First post", + "published": "2025-01-02T03:04:05Z", + "file": {"name": "clip.mp4", "path": "/data/abc/clip.mp4"}, + "attachments": [] + }"#; + let post: ApiPost = serde_json::from_str(json).expect("parse"); + let item = CoomerProvider::build_video_item(post).expect("item"); + assert_eq!(item.id, "1768608897"); + assert_eq!(item.title, "First post"); + assert_eq!( + item.url, + "https://coomer.st/onlyfans/user/tabycatxoxo/post/1768608897" + ); + assert_eq!(item.uploader.as_deref(), Some("tabycatxoxo")); + assert_eq!( + item.uploaderId.as_deref(), + Some("coomer:onlyfans:tabycatxoxo") + ); + assert_eq!(item.thumb, "https://coomer.st/data/abc/clip.mp4"); + let formats = item.formats.as_ref().expect("formats present"); + assert_eq!(formats.len(), 1); + assert_eq!(formats[0].url, "https://coomer.st/data/abc/clip.mp4"); + let headers = formats[0] + .http_headers + .as_ref() + .expect("http_headers present"); + assert_eq!(headers.get("Referer").map(String::as_str), Some(REFERER)); + assert_eq!(item.uploadedAt, Some(1735787045)); + } + + #[test] + fn build_video_item_falls_back_to_substring_title() { + let json = r#"{ + "id": "42", + "user": "ruby", + "service": "fansly", + "title": "", + "substring": "

Hello there friends

", + "file": {"name": "p.jpg", "path": "/data/x/p.jpg"} + }"#; + let post: ApiPost = serde_json::from_str(json).expect("parse"); + let item = CoomerProvider::build_video_item(post).expect("item"); + assert_eq!(item.title, "Hello there friends"); + assert_eq!(item.thumb, "https://coomer.st/data/x/p.jpg"); + // For an image attachment, no formats are emitted (preview is enough). + assert!(item.formats.is_none()); + } + + #[test] + fn build_video_item_skips_when_id_missing() { + let json = r#"{"user": "ruby", "service": "fansly"}"#; + let post: ApiPost = serde_json::from_str(json).expect("parse"); + assert!(CoomerProvider::build_video_item(post).is_none()); + } +} diff --git a/src/providers/hentaimama.rs b/src/providers/hentaimama.rs new file mode 100644 index 0000000..3c383f9 --- /dev/null +++ b/src/providers/hentaimama.rs @@ -0,0 +1,668 @@ +use crate::DbPool; +use crate::api::ClientVersion; +use crate::providers::{Provider, report_provider_error, report_provider_error_background, requester_or_default}; +use crate::status::*; +use crate::util::cache::VideoCache; +use crate::util::requester::Requester; +use crate::videos::{ServerOptions, VideoFormat, VideoItem}; +use async_trait::async_trait; +use base64::{Engine, engine::general_purpose::STANDARD}; +use chrono::DateTime; +use error_chain::error_chain; +use futures::stream::{self, StreamExt}; +use htmlentity::entity::{ICodedDataTrait, decode}; +use percent_encoding::{NON_ALPHANUMERIC, utf8_percent_encode}; +use regex::Regex; +use scraper::{ElementRef, Html, Selector}; +use std::collections::HashSet; +use std::sync::{Arc, RwLock}; +use std::thread; +use wreq::Version; + +pub const CHANNEL_METADATA: crate::providers::ProviderChannelMetadata = + crate::providers::ProviderChannelMetadata { + group_id: "hentai-animation", + tags: &["hentai", "anime", "episodes"], + }; + +error_chain! { + foreign_links { + Io(std::io::Error); + } + errors { + Parse(msg: String) { + description("parse error") + display("parse error: {}", msg) + } + } +} + +const CHANNEL_ID: &str = "hentaimama"; +const BASE_URL: &str = "https://hentaimama.io"; +// Static-file mirror the site's own "rtmp" player option points at. Serves +// direct 200s with Accept-Ranges and no Referer/token needed, so the format +// URL can be built locally from the AJAX response with no extra fetch. +const MEDIA_HOST: &str = "https://gdvid.info/"; +const USER_AGENT: &str = + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/146.0.0.0 Safari/537.36"; + +#[derive(Debug, Clone)] +enum Target { + Latest, + Genre(String), + Studio(String), + Search(String), +} + +#[derive(Debug, Clone)] +struct EpisodeCard { + id: String, + title: String, + url: String, + thumb: String, + rating: Option, +} + +#[derive(Debug, Clone)] +pub struct HentaimamaProvider { + genres: Arc>>, +} + +impl HentaimamaProvider { + pub fn new() -> Self { + let provider = Self { + genres: Arc::new(RwLock::new(vec![])), + }; + provider.spawn_initial_load(); + provider + } + + fn spawn_initial_load(&self) { + let genres = Arc::clone(&self.genres); + thread::spawn(move || { + let runtime = match tokio::runtime::Builder::new_current_thread().enable_all().build() { + Ok(runtime) => runtime, + Err(e) => { + report_provider_error_background( + CHANNEL_ID, + "spawn_initial_load.runtime_build", + &e.to_string(), + ); + return; + } + }; + runtime.block_on(async move { + if let Err(e) = Self::load_genres(genres).await { + report_provider_error_background(CHANNEL_ID, "load_genres", &e.to_string()); + } + }); + }); + } + + async fn load_genres(genres: Arc>>) -> Result<()> { + let mut requester = Requester::new(); + let html = requester + .get_with_headers( + &format!("{BASE_URL}/genres-filter/"), + Self::html_headers(BASE_URL), + Some(Version::HTTP_2), + ) + .await + .map_err(|e| Error::from(format!("genres fetch failed: {e}")))?; + + let document = Html::parse_document(&html); + let selector = Self::selector("a.genreitem")?; + let mut options = Vec::new(); + let mut seen = HashSet::new(); + for element in document.select(&selector) { + let Some(href) = element.value().attr("href") else { + continue; + }; + let slug = href.trim_end_matches('/').rsplit('/').next().unwrap_or("").to_string(); + if slug.is_empty() || !seen.insert(slug.clone()) { + continue; + } + let title = Self::decode_entities(&element.text().collect::()); + if title.is_empty() { + continue; + } + options.push(FilterOption { id: slug, title }); + } + if !options.is_empty() { + if let Ok(mut guard) = genres.write() { + *guard = options; + } + } + Ok(()) + } + + fn build_channel(&self, _clientversion: ClientVersion) -> Channel { + let genres = self.genres.read().map(|g| g.clone()).unwrap_or_default(); + Channel { + id: CHANNEL_ID.to_string(), + name: "Hentaimama".to_string(), + description: "Watch hentai episodes online free in HD.".to_string(), + premium: false, + favicon: "https://www.google.com/s2/favicons?sz=64&domain=hentaimama.io".to_string(), + status: "active".to_string(), + categories: genres.iter().map(|g| g.title.clone()).collect(), + options: vec![ChannelOption { + id: "categories".to_string(), + title: "Genres".to_string(), + description: "Filter by genre".to_string(), + systemImage: "tag.fill".to_string(), + colorName: "green".to_string(), + options: genres, + multiSelect: false, + }], + nsfw: true, + cacheDuration: Some(1800), + ytdlpCommand: Some("yt-dlp".to_string()), + } + } + + fn selector(value: &str) -> Result { + Selector::parse(value).map_err(|e| Error::from(format!("selector `{value}` parse failed: {e}"))) + } + + fn decode_entities(text: &str) -> String { + decode(text.as_bytes()) + .to_string() + .unwrap_or_else(|_| text.to_string()) + .split_whitespace() + .collect::>() + .join(" ") + .trim() + .to_string() + } + + fn html_headers(referer: &str) -> Vec<(String, String)> { + vec![ + ("Referer".to_string(), referer.to_string()), + ("User-Agent".to_string(), USER_AGENT.to_string()), + ] + } + + async fn fetch_html(requester: &mut Requester, url: &str, referer: &str) -> Result { + requester + .get_with_headers(url, Self::html_headers(referer), Some(Version::HTTP_2)) + .await + .map_err(|e| Error::from(format!("request failed for {url}: {e}"))) + } + + fn slugify(value: &str) -> String { + value + .trim() + .to_lowercase() + .chars() + .map(|c| if c.is_ascii_alphanumeric() { c } else { '-' }) + .collect::() + .split('-') + .filter(|s| !s.is_empty()) + .collect::>() + .join("-") + } + + fn find_genre_slug(&self, value: &str) -> Option { + let normalized = value.trim().to_lowercase(); + let genres = self.genres.read().ok()?; + genres + .iter() + .find(|g| g.id.eq_ignore_ascii_case(value.trim()) || g.title.to_lowercase() == normalized) + .map(|g| g.id.clone()) + } + + fn resolve_target(&self, query: Option<&str>, categories: Option<&str>) -> Target { + if let Some(value) = categories { + if let Some(slug) = self.find_genre_slug(value) { + return Target::Genre(slug); + } + let slug = Self::slugify(value); + if !slug.is_empty() { + return Target::Genre(slug); + } + } + if let Some(q) = query { + let q = q.trim(); + if let Some(rest) = q + .strip_prefix("genre:") + .or_else(|| q.strip_prefix("cat:")) + .or_else(|| q.strip_prefix("category:")) + { + let slug = self.find_genre_slug(rest).unwrap_or_else(|| Self::slugify(rest)); + return Target::Genre(slug); + } + if let Some(rest) = q.strip_prefix("studio:").or_else(|| q.strip_prefix("uploader:")) { + return Target::Studio(Self::slugify(rest)); + } + if let Some(slug) = self.find_genre_slug(q) { + return Target::Genre(slug); + } + if !q.is_empty() { + return Target::Search(q.to_string()); + } + } + Target::Latest + } + + fn build_list_url(target: &Target, page: u32) -> String { + match target { + Target::Latest => { + if page <= 1 { + format!("{BASE_URL}/episodes/") + } else { + format!("{BASE_URL}/episodes/page/{page}/") + } + } + Target::Genre(slug) => { + if page <= 1 { + format!("{BASE_URL}/genre/{slug}/") + } else { + format!("{BASE_URL}/genre/{slug}/page/{page}/") + } + } + Target::Studio(slug) => { + if page <= 1 { + format!("{BASE_URL}/studio/{slug}/") + } else { + format!("{BASE_URL}/studio/{slug}/page/{page}/") + } + } + Target::Search(query) => { + let encoded = utf8_percent_encode(query, NON_ALPHANUMERIC).to_string(); + if page <= 1 { + format!("{BASE_URL}/?s={encoded}") + } else { + format!("{BASE_URL}/page/{page}/?s={encoded}") + } + } + } + } + + fn parse_episode_cards(html: &str) -> Result> { + let document = Html::parse_document(html); + let card_selector = Self::selector("article.se.episodes")?; + let link_selector = Self::selector("div.season_m a")?; + let img_selector = Self::selector("img")?; + let rating_selector = Self::selector("div.rating")?; + let rating_re = Regex::new(r"([0-9]+(?:\.[0-9]+)?)").unwrap(); + + let mut cards = Vec::new(); + for card in document.select(&card_selector) { + let Some(id) = card.value().attr("rel").filter(|s| !s.is_empty()) else { + continue; + }; + let Some(link) = card.select(&link_selector).next() else { + continue; + }; + let Some(url) = link.value().attr("href") else { + continue; + }; + let title = Self::decode_entities(&link.text().collect::()); + if title.is_empty() { + continue; + } + let thumb = card + .select(&img_selector) + .next() + .and_then(|img| img.value().attr("src")) + .unwrap_or("") + .to_string(); + let rating = card + .select(&rating_selector) + .next() + .map(|r| r.text().collect::()) + .and_then(|text| rating_re.captures(&text).and_then(|c| c[1].parse::().ok())); + cards.push(EpisodeCard { + id: id.to_string(), + title, + url: url.to_string(), + thumb, + rating, + }); + } + Ok(cards) + } + + fn parse_series_card_urls(html: &str) -> Result> { + let document = Html::parse_document(html); + let selector = Self::selector("a.sc-poster")?; + Ok(document + .select(&selector) + .filter_map(|e| e.value().attr("href").map(|s| s.to_string())) + .collect()) + } + + fn last_episode_url(html: &str) -> Result> { + let document = Html::parse_document(html); + let selector = Self::selector("a.dt-se-item")?; + Ok(document + .select(&selector) + .filter_map(|e: ElementRef| e.value().attr("href").map(|s| s.to_string())) + .last()) + } + + fn post_id_regex() -> &'static Regex { + static RE: std::sync::OnceLock = std::sync::OnceLock::new(); + RE.get_or_init(|| Regex::new(r#"data-post-id="(\d+)""#).expect("valid regex")) + } + + fn title_regex() -> &'static Regex { + static RE: std::sync::OnceLock = std::sync::OnceLock::new(); + RE.get_or_init(|| Regex::new(r#""#).expect("valid regex")) + } + + fn thumb_regex() -> &'static Regex { + static RE: std::sync::OnceLock = std::sync::OnceLock::new(); + RE.get_or_init(|| Regex::new(r#" &'static Regex { + static RE: std::sync::OnceLock = std::sync::OnceLock::new(); + RE.get_or_init(|| Regex::new(r#"data-rating="([0-9.]+)""#).expect("valid regex")) + } + + fn published_regex() -> &'static Regex { + static RE: std::sync::OnceLock = std::sync::OnceLock::new(); + RE.get_or_init(|| { + Regex::new(r#" &'static Regex { + static RE: std::sync::OnceLock = std::sync::OnceLock::new(); + RE.get_or_init(|| { + Regex::new(r#"href="https://hentaimama\.io/(?:genre|studio)/[a-z0-9-]+/"[^>]*rel="tag">([^<]+)"#) + .expect("valid regex") + }) + } + + fn rtmp_p_regex() -> &'static Regex { + static RE: std::sync::OnceLock = std::sync::OnceLock::new(); + RE.get_or_init(|| Regex::new(r#"dt_embed=rtmp[^"]*?[?&]p=([A-Za-z0-9+/=]+)"#).expect("valid regex")) + } + + fn iframe_src_regex() -> &'static Regex { + static RE: std::sync::OnceLock = std::sync::OnceLock::new(); + RE.get_or_init(|| Regex::new(r#"src="([^"]+)""#).expect("valid regex")) + } + + fn jwplayer_file_regex() -> &'static Regex { + static RE: std::sync::OnceLock = std::sync::OnceLock::new(); + RE.get_or_init(|| Regex::new(r#"file:\s*"([^"]+)""#).expect("valid regex")) + } + + fn pad_base64(value: &str) -> String { + let padding = (4 - value.len() % 4) % 4; + format!("{value}{}", "=".repeat(padding)) + } + + /// The site's "rtmp" player option embeds the raw storage path as a base64 + /// query param; the direct mirror URL is just that path appended to + /// `MEDIA_HOST`, so no extra request is needed to resolve it. + fn format_from_p_param(iframe_html: &str) -> Option { + let captures = Self::rtmp_p_regex().captures(iframe_html)?; + let decoded = STANDARD.decode(Self::pad_base64(&captures[1])).ok()?; + let path = String::from_utf8(decoded).ok()?; + let path = path.trim().trim_start_matches('/'); + if path.is_empty() { + return None; + } + Some(VideoFormat::new( + format!("{MEDIA_HOST}{path}"), + "auto".to_string(), + "mp4".to_string(), + )) + } + + fn format_from_jwplayer(embed_html: &str) -> Option { + let src = &Self::jwplayer_file_regex().captures(embed_html)?[1]; + let container = if src.to_lowercase().contains(".m3u8") { "m3u8" } else { "mp4" }; + Some(VideoFormat::new(src.to_string(), "auto".to_string(), container.to_string())) + } + + async fn fetch_player_iframe( + requester: &mut Requester, + referer: &str, + post_id: &str, + slot: u8, + ) -> Option { + let body = format!("action=get_player_contents&a={post_id}&i={slot}"); + let headers = vec![ + ("Content-Type", "application/x-www-form-urlencoded"), + ("X-Requested-With", "XMLHttpRequest"), + ("Referer", referer), + ("User-Agent", USER_AGENT), + ]; + let response = requester + .post(&format!("{BASE_URL}/wp-admin/admin-ajax.php"), &body, headers) + .await + .ok()?; + let text = response.text().await.ok()?; + let slots: Vec = serde_json::from_str(&text).ok()?; + slots.into_iter().find(|s| !s.trim().is_empty()) + } + + async fn resolve_format(requester: &mut Requester, referer: &str, post_id: &str) -> Option { + let iframe_html = Self::fetch_player_iframe(requester, referer, post_id, 1).await?; + if let Some(format) = Self::format_from_p_param(&iframe_html) { + return Some(format); + } + // Fallback for embed shapes we haven't seen: follow the iframe itself and + // pull the jwplayer source out of its markup. + let src = &Self::iframe_src_regex().captures(&iframe_html)?[1]; + let embed_html = Self::fetch_html(requester, src, referer).await.ok()?; + Self::format_from_jwplayer(&embed_html) + } + + async fn build_item_from_episode_card(requester: &mut Requester, entry: &EpisodeCard) -> Option { + let format = Self::resolve_format(requester, &entry.url, &entry.id).await?; + let mut item = VideoItem::new( + entry.id.clone(), + entry.title.clone(), + entry.url.clone(), + CHANNEL_ID.to_string(), + entry.thumb.clone(), + 0, + ) + .formats(vec![format]) + .aspect_ratio(16.0 / 9.0); + if let Some(rating) = entry.rating { + item = item.rating(rating * 10.0); + } + Some(item) + } + + /// Search/genre/studio archives only expose series (show) cards, so a match + /// is represented by its most recent episode - the same shape the latest + /// feed already returns, just reached through one extra hop. + async fn resolve_series_to_item(requester: &mut Requester, series_url: &str) -> Option { + let html = Self::fetch_html(requester, series_url, BASE_URL).await.ok()?; + let episode_url = Self::last_episode_url(&html).ok().flatten()?; + Self::resolve_episode_url(requester, &episode_url).await + } + + async fn resolve_episode_url(requester: &mut Requester, episode_url: &str) -> Option { + let html = Self::fetch_html(requester, episode_url, BASE_URL).await.ok()?; + let post_id = Self::post_id_regex().captures(&html)?[1].to_string(); + let title = Self::decode_entities(&Self::title_regex().captures(&html)?[1]); + if title.is_empty() { + return None; + } + let thumb = Self::thumb_regex() + .captures(&html) + .map(|c| c[1].to_string()) + .unwrap_or_default(); + let rating = Self::rating_regex().captures(&html).and_then(|c| c[1].parse::().ok()); + let uploaded_at = Self::published_regex() + .captures(&html) + .and_then(|c| DateTime::parse_from_rfc3339(&c[1]).ok()) + .map(|dt| dt.timestamp().max(0) as u64); + let mut seen = HashSet::new(); + let tags: Vec = Self::tag_regex() + .captures_iter(&html) + .map(|c| Self::decode_entities(&c[1])) + .filter(|t| !t.is_empty() && seen.insert(t.clone())) + .collect(); + + let format = Self::resolve_format(requester, episode_url, &post_id).await?; + + let mut item = VideoItem::new( + post_id, + title, + episode_url.to_string(), + CHANNEL_ID.to_string(), + thumb, + 0, + ) + .formats(vec![format]) + .tags(tags) + .aspect_ratio(16.0 / 9.0); + if let Some(rating) = rating { + item = item.rating(rating * 10.0); + } + if let Some(uploaded_at) = uploaded_at { + item = item.uploaded_at(uploaded_at); + } + Some(item) + } + + async fn get( + &self, + cache: VideoCache, + page: u32, + per_page: usize, + query: Option<&str>, + options: ServerOptions, + ) -> Result> { + let target = self.resolve_target(query, options.categories.as_deref()); + let list_url = Self::build_list_url(&target, page); + + if let Some((time, items)) = cache.get(&list_url) { + if time.elapsed().unwrap_or_default().as_secs() < 300 { + return Ok(items); + } + } + + let mut requester = requester_or_default(&options, CHANNEL_ID, "get_videos"); + let html = Self::fetch_html(&mut requester, &list_url, BASE_URL).await?; + + let items = match target { + Target::Latest => { + let cards = Self::parse_episode_cards(&html)?; + stream::iter(cards.into_iter().take(per_page.max(1)).map(|entry| { + let mut req = requester.clone(); + async move { Self::build_item_from_episode_card(&mut req, &entry).await } + })) + .buffer_unordered(6) + .filter_map(|item| async move { item }) + .collect::>() + .await + } + _ => { + let urls = Self::parse_series_card_urls(&html)?; + stream::iter(urls.into_iter().take(per_page.max(1)).map(|url| { + let mut req = requester.clone(); + async move { Self::resolve_series_to_item(&mut req, &url).await } + })) + .buffer_unordered(4) + .filter_map(|item| async move { item }) + .collect::>() + .await + } + }; + + if !items.is_empty() { + cache.insert(list_url, items.clone()); + } + Ok(items) + } +} + +#[async_trait] +impl Provider for HentaimamaProvider { + async fn get_videos( + &self, + cache: VideoCache, + _pool: DbPool, + _sort: String, + query: Option, + page: String, + per_page: String, + options: ServerOptions, + ) -> Vec { + let page = page.parse::().unwrap_or(1).max(1); + let per_page = per_page.parse::().unwrap_or(24); + let query_ref = query.as_deref().filter(|q| !q.trim().is_empty()); + + match self.get(cache, page, per_page, query_ref, options).await { + Ok(items) => items, + Err(e) => { + report_provider_error(CHANNEL_ID, "get_videos", &e.to_string()).await; + vec![] + } + } + } + + fn get_channel(&self, clientversion: ClientVersion) -> Option { + Some(self.build_channel(clientversion)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn provider() -> HentaimamaProvider { + HentaimamaProvider { + genres: Arc::new(RwLock::new(vec![FilterOption { + id: "maid".to_string(), + title: "Maid".to_string(), + }])), + } + } + + #[test] + fn builds_latest_urls() { + assert_eq!( + HentaimamaProvider::build_list_url(&Target::Latest, 1), + "https://hentaimama.io/episodes/" + ); + assert_eq!( + HentaimamaProvider::build_list_url(&Target::Latest, 2), + "https://hentaimama.io/episodes/page/2/" + ); + } + + #[test] + fn builds_genre_and_search_urls() { + assert_eq!( + HentaimamaProvider::build_list_url(&Target::Genre("maid".to_string()), 1), + "https://hentaimama.io/genre/maid/" + ); + assert_eq!( + HentaimamaProvider::build_list_url(&Target::Search("school girl".to_string()), 2), + "https://hentaimama.io/page/2/?s=school%20girl" + ); + } + + #[test] + fn routes_query_shortcuts() { + let provider = provider(); + assert!(matches!(provider.resolve_target(Some("maid"), None), Target::Genre(s) if s == "maid")); + assert!(matches!(provider.resolve_target(Some("genre:blowjob"), None), Target::Genre(s) if s == "blowjob")); + assert!(matches!(provider.resolve_target(Some("studio:majin"), None), Target::Studio(s) if s == "majin")); + assert!(matches!(provider.resolve_target(Some("random keyword"), None), Target::Search(s) if s == "random keyword")); + assert!(matches!(provider.resolve_target(None, None), Target::Latest)); + } + + #[test] + fn decodes_gdvid_path_from_p_param() { + let iframe = r#""#; + let format = HentaimamaProvider::format_from_p_param(iframe).unwrap(); + assert_eq!( + format.url, + "https://gdvid.info/P/pure-x-holic-junketsu-otome-to-konin-kankei-the-animation-2.mp4" + ); + } +}