8 Commits
Author SHA1 Message Date
SalastilandClaude Sonnet 5.5 f3f872349a nobilis 1.0.0-rc.5
Co-Authored-By: Claude Sonnet 5.5 <[email protected]>
2026-10-10 16:36:55 -04:00
SalastilandClaude Sonnet 5.5 c4b5043e0b Local debug builds keep line numbers, not full debug information
Co-Authored-By: Claude Sonnet 5.5 <[email protected]>
2026-10-10 16:12:52 -04:00
SalastilandClaude Sonnet 5.5 b0fa250854 Muting a Discord channel or direct message sets it on the account, so it is quiet on a phone too
Co-Authored-By: Claude Sonnet 5.5 <[email protected]>
2026-10-10 13:33:05 -04:00
SalastilandClaude Sonnet 5.5 dab394fc99 Link cards on every service, every way a message arrives: an edit keeps the cards whose link is still in it and describes one it added, and a conversation read describes the links stored without one - Matrix history, Discord catch-up, older scrollback
Co-Authored-By: Claude Sonnet 5.5 <[email protected]>
2026-10-09 23:07:05 -04:00
SalastilandClaude Sonnet 5.5 340342b261 A YouTube card over Tor: oEmbed answers every Tor exit with a 403, so when it is refused the video's own page is read for the title and channel instead
Co-Authored-By: Claude Sonnet 5.5 <[email protected]>
2026-10-09 22:57:05 -04:00
SalastilandClaude Sonnet 5.5 ca29a63883 Sneedchat: the forum's direct file links - AVIF pictures and MP4s from its uploads host - have their domain taken off and are written for the account, then fetched through its own route and kept, as the short attachment form is; the [img] tags go with the link
Co-Authored-By: Claude Sonnet 5.5 <[email protected]>
2026-10-09 19:06:23 -04:00
SalastilandClaude Sonnet 5.5 046fbb4850 GIFs: Discord's gifv embeds are kept as the GIFs they are, Tenor and Giphy links are resolved for every other service, and the picker's list comes from a Discord account (#331)
Co-Authored-By: Claude Sonnet 5.5 <[email protected]>
2026-10-09 17:55:10 -04:00
SalastilandClaude Sonnet 5.5 35ea0b7be9 Idle when the person steps away: setUserIdle moves an online account on a service with an idle to idle and back, as session presence only; what somebody chose is never moved, nor anyone in a call (#332)
Co-Authored-By: Claude Sonnet 5.5 <[email protected]>
2026-10-09 17:35:58 -04:00
16 changed files with 1333 additions and 31 deletions
Generated
+1 -1
View File
@@ -3680,7 +3680,7 @@ checksum = "650eef8c711430f1a879fdd01d4745a7deea475becfb90269c06775983bbf086"
[[package]]
name = "nobilis"
version = "1.0.0-rc.4"
version = "1.0.0-rc.5"
dependencies = [
"aes-gcm",
"anyhow",
+7 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "nobilis"
version = "1.0.0-rc.4"
version = "1.0.0-rc.5"
edition = "2021"
description = "Chat daemon: IRC, Discord, Sneedchat and Matrix behind one unified JSON model over a Unix socket"
license = "GPL-3.0-or-later"
@@ -130,3 +130,9 @@ libc = "0.2"
# which it has. Reported upstream; drop this once a release carries the fix.
[patch.crates-io]
saturating-time = { path = "vendor/saturating-time" }
# Local debug builds keep line numbers for backtraces and not the full debug
# information, which made each one several times the size of the program. A
# release build, which is what ships, is not affected.
[profile.dev]
debug = "line-tables-only"
+5
View File
@@ -1083,6 +1083,7 @@ pub fn irc_account_to_json(a: &IrcAccountConfig, state: &str) -> Account {
Account {
status: "online".to_string(),
status_text: String::new(),
auto_idle: false,
display_name: a.display_name.clone().filter(|s| !s.is_empty()).unwrap_or_else(|| id.clone()),
id,
service: "irc".to_string(),
@@ -1125,6 +1126,7 @@ pub fn discord_account_to_json(a: &DiscordAccountConfig, state: &str) -> Account
service: "discord".to_string(),
status: "online".to_string(),
status_text: String::new(),
auto_idle: false,
display_name: a.display_name.clone().filter(|s| !s.is_empty()).unwrap_or_else(|| a.username.clone()),
state: state.to_string(),
autojoin: String::new(),
@@ -1162,6 +1164,7 @@ pub fn sneedchat_account_to_json(a: &SneedChatAccountConfig, state: &str) -> Acc
service: "sneedchat".to_string(),
status: "online".to_string(),
status_text: String::new(),
auto_idle: false,
display_name: a.display_name.clone().filter(|s| !s.is_empty()).unwrap_or_else(|| a.username.clone()),
state: state.to_string(),
autojoin: String::new(),
@@ -1204,6 +1207,7 @@ pub fn kick_account_to_json(a: &KickAccountConfig, state: &str) -> Account {
service: "kick".to_string(),
status: "online".to_string(),
status_text: String::new(),
auto_idle: false,
display_name: a.display_name.clone().filter(|s| !s.is_empty()).unwrap_or_else(|| a.username.clone()),
state: state.to_string(),
// The watched channels are deliberately not reported as `autojoin`.
@@ -1249,6 +1253,7 @@ pub fn matrix_account_to_json(a: &MatrixAccountConfig, state: &str, has_key_back
service: "matrix".to_string(),
status: "online".to_string(),
status_text: String::new(),
auto_idle: false,
display_name: a.display_name.clone().filter(|s| !s.is_empty()).unwrap_or_else(|| a.user_id.clone()),
state: state.to_string(),
autojoin: String::new(),
+1 -1
View File
@@ -324,7 +324,7 @@ pub(super) async fn run_gateway(state: &AppState, config: &DiscordAccountConfig,
"large_threshold": 50,
// Carried in IDENTIFY as well as pushed live, so a status set
// before a reconnect survives it.
"presence": presence_payload(&state.runtime.account_status(&account_id), &state.runtime.account_status_text(&account_id)),
"presence": presence_payload(&state.runtime.effective_status(state, &account_id), &state.runtime.account_status_text(&account_id)),
}
})
.to_string(),
+164
View File
@@ -0,0 +1,164 @@
//! The GIFs Discord offers, for browsing.
//!
//! Discord's own client gets its GIFs from the same two requests - what is
//! trending, and a search - and shows what comes back; this asks for the same
//! and does no more than the official client does: reads, and only when the
//! picker is open and somebody has stopped typing.
//!
//! Any Discord account will do, whichever conversation the picker is open in,
//! since what it borrows is the account's way of asking and not the account's
//! conversations: a GIF found this way can be sent anywhere.
//!
//! Answers are kept for a few minutes. Opening the picker twice, or typing a
//! word and deleting it, is not a reason to ask again.
use super::*;
use std::sync::LazyLock;
use std::time::Instant;
static CACHE: LazyLock<std::sync::Mutex<HashMap<String, (Instant, Value)>>> = LazyLock::new(|| std::sync::Mutex::new(HashMap::new()));
const TRENDING_FOR: Duration = Duration::from_secs(10 * 60);
const SEARCH_FOR: Duration = Duration::from_secs(5 * 60);
const CACHE_LIMIT: usize = 64;
/// One GIF as the picker wants it: where it is, what it is called, and what to
/// send. Discord's own words for these are `src` for the video, `gif_src` for
/// the file, `url` for the page - kept, since each is used for something.
fn one(item: &Value) -> Option<Value> {
let url = item["url"].as_str().filter(|u| !u.is_empty())?;
let src = item["src"].as_str().filter(|u| !u.is_empty());
let gif = item["gif_src"].as_str().filter(|u| !u.is_empty());
// Something to play, or there is nothing to show.
let play = src.or(gif)?;
Some(json!({
"id": item["id"].as_str().map(str::to_string).unwrap_or_else(|| url.to_string()),
"title": item["title"].as_str().unwrap_or(""),
"url": url,
"src": play,
"gifSrc": gif.unwrap_or(play),
"preview": item["preview"].as_str().unwrap_or(""),
"width": item["width"].as_u64().unwrap_or(0),
"height": item["height"].as_u64().unwrap_or(0),
}))
}
/// Discord's two shapes of answer: a search is a list of GIFs, and trending is
/// an object holding them with the categories beside.
pub(super) fn parse_gifs(answer: &Value) -> Value {
let list = answer.as_array().or_else(|| answer["gifs"].as_array()).or_else(|| answer["results"].as_array());
let gifs: Vec<Value> = list.into_iter().flatten().filter_map(one).collect();
let categories: Vec<Value> = answer["categories"]
.as_array()
.into_iter()
.flatten()
.filter_map(|c| {
let name = c["name"].as_str().filter(|n| !n.is_empty())?;
Some(json!({ "name": name, "src": c["src"].as_str().unwrap_or("") }))
})
.collect();
json!({ "gifs": gifs, "categories": categories })
}
/// What is trending, or what matches `query`.
pub async fn list_gifs(state: &AppState, query: &str) -> Result<Value> {
let query = query.trim();
let key = query.to_lowercase();
let ttl = if key.is_empty() { TRENDING_FOR } else { SEARCH_FOR };
if let Some((at, answer)) = CACHE.lock().unwrap().get(&key) {
if at.elapsed() < ttl {
return Ok(answer.clone());
}
}
let config = state
.accounts
.all_discord()
.into_iter()
.next()
.context("GIFs are found through a Discord account, and none is set up")?;
let mut url = url::Url::parse(&format!("{API_BASE}/gifs/{}", if query.is_empty() { "trending" } else { "search" })).context("building the GIF request")?;
{
let mut pairs = url.query_pairs_mut();
if !query.is_empty() {
pairs.append_pair("q", &query.chars().take(100).collect::<String>());
}
pairs.append_pair("media_format", "mp4").append_pair("locale", "en-US");
}
let resp = http_client_for(&config.token)
.get(url)
.header("Authorization", &config.token)
.send()
.await
.context("asking Discord for GIFs")?;
if !resp.status().is_success() {
let status = resp.status();
let text = resp.text().await.unwrap_or_default();
bail!("{}", discord_error_text(status, &text, "asking Discord for GIFs"));
}
let answer: Value = resp.json().await.context("reading Discord's GIFs")?;
let parsed = parse_gifs(&answer);
let mut cache = CACHE.lock().unwrap();
if cache.len() >= CACHE_LIMIT {
cache.clear();
}
cache.insert(key, (Instant::now(), parsed.clone()));
Ok(parsed)
}
#[cfg(test)]
mod tests {
use super::*;
fn gif(id: &str) -> Value {
json!({
"id": id, "title": "cat", "url": format!("https://tenor.com/view/cat-gif-{id}"),
"src": format!("https://media.tenor.com/{id}.mp4"), "gif_src": format!("https://media.tenor.com/{id}.gif"),
"preview": format!("https://media.tenor.com/{id}.png"), "width": 480, "height": 270
})
}
#[test]
fn a_search_is_a_list_of_gifs() {
let out = parse_gifs(&json!([gif("1"), gif("2")]));
assert_eq!(out["gifs"].as_array().unwrap().len(), 2);
assert_eq!(out["gifs"][0]["url"], "https://tenor.com/view/cat-gif-1");
assert_eq!(out["gifs"][0]["src"], "https://media.tenor.com/1.mp4");
assert_eq!(out["gifs"][0]["gifSrc"], "https://media.tenor.com/1.gif");
assert_eq!((out["gifs"][0]["width"].as_u64(), out["gifs"][0]["height"].as_u64()), (Some(480), Some(270)));
}
#[test]
fn trending_holds_its_gifs_beside_its_categories() {
let out = parse_gifs(&json!({
"categories": [{ "name": "reaction", "src": "https://media.tenor.com/c.mp4" }, { "name": "" }],
"gifs": [gif("1")]
}));
assert_eq!(out["gifs"].as_array().unwrap().len(), 1);
assert_eq!(out["categories"].as_array().unwrap().len(), 1, "a category with no name is not one");
assert_eq!(out["categories"][0]["name"], "reaction");
}
#[test]
fn one_with_nothing_to_play_or_no_page_is_left_out() {
let out = parse_gifs(&json!([
{ "url": "https://tenor.com/view/x" },
{ "src": "https://media.tenor.com/y.mp4" },
gif("3")
]));
assert_eq!(out["gifs"].as_array().unwrap().len(), 1);
}
#[test]
fn without_a_file_the_video_is_what_there_is() {
let out = parse_gifs(&json!([{ "url": "https://tenor.com/view/x", "src": "https://media.tenor.com/x.mp4" }]));
assert_eq!(out["gifs"][0]["gifSrc"], "https://media.tenor.com/x.mp4");
}
#[test]
fn nothing_recognisable_is_nothing() {
let out = parse_gifs(&json!({ "message": "nope" }));
assert!(out["gifs"].as_array().unwrap().is_empty());
assert!(out["categories"].as_array().unwrap().is_empty());
}
}
+104
View File
@@ -187,6 +187,12 @@ pub(super) fn extract_body(d: &Value) -> Option<String> {
continue;
}
}
// A GIF is drawn from its embed. Passing its media address on as
// well would be the same picture twice, and was how a GIF whose
// link was not in the text showed as a bare mp4 address.
if embed["type"].as_str() == Some("gifv") {
continue;
}
if let Some(url) = embed["video"]["url"].as_str() {
parts.push(url.to_string());
} else if let Some(url) = embed["image"]["url"].as_str() {
@@ -704,6 +710,9 @@ pub(super) fn extract_embeds(d: &Value) -> Vec<Embed> {
if embed["type"].as_str() == Some("safety_system_notification") {
return Some(safety_notice(embed));
}
if embed["type"].as_str() == Some("gifv") {
return gif_embed(embed);
}
let title = embed["title"].as_str().filter(|s| !s.is_empty()).map(|s| s.to_string());
let description = embed["description"].as_str().filter(|s| !s.is_empty()).map(|s| s.to_string());
let fields: Vec<model::EmbedField> = embed["fields"]
@@ -742,6 +751,41 @@ pub(super) fn extract_embeds(d: &Value) -> Vec<Embed> {
.collect()
}
/// A GIF from Tenor, Klipy, Giphy or whoever Discord unfurled it from.
///
/// Discord answers a link to one with an embed of type `gifv` whose `video` is
/// the moving picture itself - an mp4 - and whose `url` is the page that was
/// linked, which has no file extension to recognise it by. That is why a
/// posted GIF used to show as nothing but its address.
///
/// The proxied copy is preferred where there is one: it is served from
/// Discord's own CDN, so reading a message does not tell a third party's server
/// who read it, which is the same arrangement Discord's own client makes.
fn gif_embed(embed: &Value) -> Option<Embed> {
let video = &embed["video"];
let from_video = video["proxy_url"].as_str().or_else(|| video["url"].as_str()).filter(|u| !u.is_empty());
let media = from_video
.or_else(|| embed["thumbnail"]["proxy_url"].as_str().filter(|u| !u.is_empty()))
.or_else(|| embed["thumbnail"]["url"].as_str().filter(|u| !u.is_empty()))?;
let size = |key: &str| {
video[key]
.as_u64()
.or_else(|| embed["thumbnail"][key].as_u64())
.and_then(|n| u32::try_from(n).ok())
.filter(|&n| n > 0)
};
Some(Embed {
kind: Some("gif".to_string()),
url: embed["url"].as_str().map(str::to_string),
media_url: Some(media.to_string()),
media_kind: Some(if from_video.is_some() { "video" } else { "image" }.to_string()),
width: size("width"),
height: size("height"),
provider: embed["provider"]["name"].as_str().filter(|s| !s.is_empty()).map(str::to_string),
..Default::default()
})
}
/// One of the platform's own notices to an account - a warning, a removed
/// violation, a limit on the account - which arrive as an embed of this type
/// whose content is a list of labelled values: `header`, `body`, `icon_type`,
@@ -1620,3 +1664,63 @@ mod safety_notice_tests {
assert!(cta.url.is_none());
}
}
#[cfg(test)]
mod gif_tests {
use super::{extract_body, extract_embeds};
use serde_json::json;
/// A Tenor GIF as Discord unfurls it.
fn gifv() -> serde_json::Value {
json!({
"id": "1", "channel_id": "2", "content": "https://tenor.com/view/rickroll-gif-22954713",
"embeds": [{
"type": "gifv", "url": "https://tenor.com/view/rickroll-gif-22954713",
"provider": { "name": "Tenor", "url": "https://tenor.co" },
"thumbnail": { "url": "https://media.tenor.com/x.png", "width": 640, "height": 640 },
"video": { "url": "https://media.tenor.com/x.mp4", "proxy_url": "https://images-ext-1.discordapp.net/external/x.mp4", "width": 640, "height": 640 }
}]
})
}
#[test]
fn a_gifv_is_a_gif_with_somewhere_to_play_it() {
let e = extract_embeds(&gifv());
assert_eq!(e.len(), 1);
assert_eq!(e[0].kind.as_deref(), Some("gif"));
assert_eq!(e[0].url.as_deref(), Some("https://tenor.com/view/rickroll-gif-22954713"));
assert_eq!((e[0].width, e[0].height), (Some(640), Some(640)));
assert_eq!(e[0].provider.as_deref(), Some("Tenor"));
assert_eq!(e[0].media_kind.as_deref(), Some("video"));
}
#[test]
fn a_gifv_with_only_a_thumbnail_is_a_picture() {
let d = json!({ "embeds": [{ "type": "gifv", "url": "https://x/y", "thumbnail": { "url": "https://x/t.png" } }] });
assert_eq!(extract_embeds(&d)[0].media_kind.as_deref(), Some("image"));
}
#[test]
fn it_is_played_from_discords_own_copy_so_the_third_party_is_not_told() {
let e = extract_embeds(&gifv());
assert_eq!(e[0].media_url.as_deref(), Some("https://images-ext-1.discordapp.net/external/x.mp4"));
let mut direct = gifv();
direct["embeds"][0]["video"].as_object_mut().unwrap().remove("proxy_url");
assert_eq!(extract_embeds(&direct)[0].media_url.as_deref(), Some("https://media.tenor.com/x.mp4"));
}
#[test]
fn a_gifv_with_no_picture_anywhere_is_not_one() {
let d = json!({ "embeds": [{ "type": "gifv", "url": "https://tenor.com/view/x" }] });
assert!(extract_embeds(&d).is_empty());
}
#[test]
fn the_picture_is_not_also_added_to_the_words() {
// Whether or not the link is in the text: the embed is what carries it.
assert_eq!(extract_body(&gifv()).as_deref(), Some("https://tenor.com/view/rickroll-gif-22954713"));
let mut quiet = gifv();
quiet["content"] = json!("");
assert_eq!(extract_body(&quiet), None);
}
}
+2
View File
@@ -73,10 +73,12 @@ pub mod stickers;
pub mod soundboard;
pub mod events;
pub mod forums;
pub mod gifpicker;
pub use calls::*;
pub use commands::*;
pub use forums::*;
pub use gifpicker::*;
pub use gateway::*;
pub use guilds::*;
pub use history::*;
+49 -3
View File
@@ -7,9 +7,10 @@
//! months ago was as loud in moho as anything else - and the busiest servers
//! are exactly the ones people mute.
//!
//! Reading it, not writing it. Being *told* is the half that matters and the
//! half that can go wrong quietly; setting it from here is a separate thing
//! and can follow.
//! Read first, and then written for a channel: muting one from here sets it on
//! the account, so it is quiet on a phone as well as in this window - which
//! is what somebody means by muting it. Discord answers with the account's
//! settings as they now stand, and those are taken in like any others.
//!
//! Applying is deliberately its own step. The settings arrive in READY, and
//! the channels they name arrive afterwards in GUILD_CREATE - so a mute
@@ -81,6 +82,42 @@ pub(super) fn note_settings(state: &AppState, account_id: &str, settings: &Value
apply(state, account_id);
}
/// What to send to mute or unmute one channel: its own setting inside its
/// guild's, with no end time - a mute that does not expire.
fn override_body(channel_id: &str, muted: bool) -> Value {
json!({ "channel_overrides": { channel_id: { "muted": muted, "mute_config": null } } })
}
/// Mutes or unmutes a conversation on the account itself.
///
/// A muted channel still counts what was said to its reader - Discord keeps
/// the mention badge on one - and stops sounding on every device signed in to
/// the account. Direct messages are the `@me` scope.
pub async fn set_muted(state: &AppState, account_id: &str, buffer_id: &str, muted: bool) -> anyhow::Result<()> {
use anyhow::Context;
let config = state.accounts.get_discord(account_id).context("no such account")?;
let channel_id = state.runtime.get_discord_channel(buffer_id).context("not a Discord conversation")?;
let guild = state.runtime.get_discord_guild(buffer_id).unwrap_or_else(|| DM_SCOPE.to_string());
let response = send_write(
http_client_for(&config.token)
.patch(format!("{API_BASE}/users/@me/guilds/{guild}/settings"))
.header("Authorization", &config.token)
.json(&override_body(&channel_id, muted)),
)
.await?;
let status = response.status();
if !status.is_success() {
anyhow::bail!("Discord refused to {} it (HTTP {})", if muted { "mute" } else { "unmute" }, status.as_u16());
}
// The account's settings as they now are. The gateway says the same a
// moment later; taking it here as well is what makes the change show
// without waiting on that.
if let Ok(now) = response.json::<Value>().await {
note_settings(state, account_id, &now);
}
Ok(())
}
/// Puts what is known onto the buffers that exist.
///
/// Cheap enough to run whenever anything might have changed - one pass over
@@ -112,6 +149,15 @@ mod tests {
const NOW: i64 = 1_800_000_000;
#[test]
fn muting_a_channel_is_its_own_setting_with_no_end() {
assert_eq!(
override_body("100", true),
json!({ "channel_overrides": { "100": { "muted": true, "mute_config": null } } })
);
assert_eq!(override_body("100", false)["channel_overrides"]["100"]["muted"], false);
}
/// A muted guild and, inside it, one channel muted on its own. Both are
/// read; neither is inferred from the other.
#[test]
+13
View File
@@ -85,6 +85,19 @@ pub async fn apply_status(state: &AppState, account_id: &str, status: &str, text
}
}
/// Shows the account as `status` on this session and leaves the saved setting
/// alone.
///
/// For being idle because the person has walked away, which is a fact about
/// this session and not a choice: Discord's own client does the same, and
/// writing it to the account's settings would turn a coffee break into the
/// status every other device starts from.
pub fn apply_session_presence(state: &AppState, account_id: &str, status: &str) -> bool {
let Some(sender) = state.runtime.discord_gateway_sender(account_id) else { return false };
let said = state.runtime.account_status_text(account_id);
sender.send(json!({ "op": 3, "d": presence_payload(status, &said) }).to_string()).is_ok()
}
/// What goes to Discord's settings: the status, and the custom status only when
/// it is being changed. Null clears it; an object sets it.
fn settings_body(status: &str, text: Option<&str>) -> serde_json::Value {
+312 -12
View File
@@ -382,6 +382,112 @@ pub(super) fn find_attachment_url(text: &str) -> Option<(&str, &str, &str)> {
pub const KIWIFARMS_CLEARNET_HOST: &str = "kiwifarms.st";
/// The forum's direct file links: what an `[img]` tag or a pasted address holds
/// when the picture is served from its uploads host rather than through an
/// attachment page.
///
/// `https://uploads.kiwifarms.st/data/attachments/9671/9671535-<hash>-alt.avif?hash=...`
/// for a picture - the `-alt` is XenForo's AVIF copy of one - and
/// `.../data/video/9667/9667943-<hash>.mp4?hash=...` for a video. The `hash` is
/// what the host asks for in place of a login, so the address is fetched with it
/// and as written.
///
/// Recognised so they can be fetched the way an attachment page is: through the
/// account's own route, to a file kept here. Left to the window they are asked
/// for from the reader's own address, which the host turns away and which, for
/// somebody on Tor, is the leak routing was meant to prevent.
///
/// Returns the matched substring, the attachment's id and extension, and the
/// path and query to ask the forum for.
pub(super) fn find_upload_url(text: &str) -> Option<(&str, String, String, String)> {
let mut best: Option<(&str, String, String, String)> = None;
for prefix in [
format!("https://uploads.{KIWIFARMS_CLEARNET_HOST}/data/"),
format!("https://{DEFAULT_ONION}/data/"),
format!("http://{DEFAULT_ONION}/data/"),
] {
let Some(start) = text.find(&prefix) else { continue };
let tail_start = start + prefix.len();
let rest = &text[tail_start..];
let len = rest.find(|c: char| c.is_whitespace() || matches!(c, '<' | '[' | ']' | '"' | '\'')).unwrap_or(rest.len());
let tail = &rest[..len];
let (path, query) = tail.split_once('?').map_or((tail, ""), |(p, q)| (p, q));
// `attachments/<bucket>/<id>-<hash>[-alt].<ext>` or `video/<bucket>/...`.
let mut parts = path.split('/');
let (Some(kind), Some(bucket), Some(file), None) = (parts.next(), parts.next(), parts.next(), parts.next()) else { continue };
if !matches!(kind, "attachments" | "video") || bucket.is_empty() || !bucket.bytes().all(|b| b.is_ascii_digit()) {
continue;
}
let Some((stem, ext)) = file.rsplit_once('.') else { continue };
let ext = ext.to_ascii_lowercase();
if !ATTACHMENT_EXTS.contains(&ext.as_str()) {
continue;
}
let id: String = stem.chars().take_while(|c| c.is_ascii_digit()).collect();
if id.is_empty() || !stem[id.len()..].starts_with('-') {
continue;
}
let matched_end = tail_start + path.len() + if query.is_empty() { 0 } else { 1 + query.len() };
let ask = format!("/data/{path}{}", if query.is_empty() { String::new() } else { format!("?{query}") });
if best.as_ref().is_none_or(|(m, ..)| start < text.find(m).unwrap_or(usize::MAX)) {
best = Some((&text[start..matched_end], id, ext, ask));
}
}
best
}
/// What a message links to that is fetched and kept for the window, by either
/// kind of link, and where to ask for it.
struct Resolvable {
/// The link as written, to be taken out of the text once the file is kept.
matched: String,
id: String,
ext: String,
fetch_url: String,
/// Where the same thing is on the open internet, for "open in browser".
page_url: Option<String>,
}
/// The earliest link in `body` of either kind, with where this account asks
/// for it: the address it reaches the forum at, which through Tor is the onion
/// address, and on the open internet is the forum's own - its uploads host for
/// a direct file link.
fn find_resolvable(body: &str, host: &str) -> Option<Resolvable> {
let page = find_attachment_url(body).and_then(|(matched, id, ext)| {
let at = matched.find("/attachments/")?;
Some((
body.find(matched)?,
Resolvable {
matched: matched.to_string(),
id: id.to_string(),
ext: ext.to_ascii_lowercase(),
fetch_url: format!("https://{host}{}", &matched[at..]),
page_url: Some(format!("https://{KIWIFARMS_CLEARNET_HOST}{}", &matched[at..])),
},
))
});
let upload = find_upload_url(body).and_then(|(matched, id, ext, ask)| {
let fetch_host = upload_host(host);
Some((
body.find(matched)?,
Resolvable { matched: matched.to_string(), id, ext, fetch_url: format!("https://{fetch_host}{ask}"), page_url: Some(matched.to_string()) },
))
});
match (page, upload) {
(Some((a, page)), Some((b, upload))) => Some(if a <= b { page } else { upload }),
(Some((_, page)), None) => Some(page),
(None, Some((_, upload))) => Some(upload),
(None, None) => None,
}
}
/// The text with a link taken out, and the `[img]` tags that held it with it:
/// left behind they are an empty tag, which draws as nothing and still takes a
/// line.
fn without_link(body: &str, link: &str) -> String {
body.replace(&format!("[img]{link}[/img]"), "").replace(link, "").trim().to_string()
}
/// Writes the forum's short attachment form out in full.
///
/// People type only the path - `/attachments/seal-png.9609686/` - and the
@@ -396,6 +502,89 @@ pub const KIWIFARMS_CLEARNET_HOST: &str = "kiwifarms.st";
/// with a media extension) - so `https://example.com/attachments/...`, or a
/// path somebody is merely talking about, is left as it is.
pub(super) fn expand_attachment_paths<'a>(body: &'a str, host: &str) -> std::borrow::Cow<'a, str> {
let pages = expand_attachment_pages(body, host);
match expand_data_links(&pages, host) {
std::borrow::Cow::Borrowed(_) => pages,
std::borrow::Cow::Owned(done) => std::borrow::Cow::Owned(done),
}
}
/// The address a direct file link is asked for at: the onion address through
/// Tor, and on the open internet the forum's uploads host, which is where its
/// files are served from.
fn upload_host(host: &str) -> String {
if host == KIWIFARMS_CLEARNET_HOST { format!("uploads.{KIWIFARMS_CLEARNET_HOST}") } else { host.to_string() }
}
/// The forum's direct file links - `/data/attachments/...` and `/data/video/...`
/// - written for this account, whatever way they arrived.
///
/// Said bare, they are given the address this account reaches the forum at,
/// as `/attachments/...` is. Said with a domain - the uploads host, the
/// forum's own, or its onion - the domain is taken off first and the same is
/// done, so a link copied out of a browser is the same thing as one typed
/// short: it goes through the account's own route, and the address of a site
/// the account is routed away from never reaches the window at all. Written
/// with the domain left on, the window asked the clearnet host for the picture
/// from the reader's own address, which it refuses and which is the leak that
/// routing exists to prevent.
///
/// Only a link shaped like one of the forum's files, on one of the forum's
/// own addresses (or with none), so a link to anywhere else - or a path
/// somebody is merely talking about - is left as it is.
fn expand_data_links<'a>(body: &'a str, host: &str) -> std::borrow::Cow<'a, str> {
if !body.contains("/data/") {
return std::borrow::Cow::Borrowed(body);
}
let origins: [String; 6] = [
format!("https://uploads.{KIWIFARMS_CLEARNET_HOST}"),
format!("https://{KIWIFARMS_CLEARNET_HOST}"),
format!("http://{KIWIFARMS_CLEARNET_HOST}"),
format!("https://www.{KIWIFARMS_CLEARNET_HOST}"),
format!("https://{DEFAULT_ONION}"),
format!("http://{DEFAULT_ONION}"),
];
let mut out = String::with_capacity(body.len() + 80);
let mut cursor = 0usize;
let mut search = 0usize;
let mut changed = false;
while let Some(found) = body[search..].find("/data/") {
let at = search + found;
search = at + "/data/".len();
let tail = &body[at..];
let len = tail.find(|c: char| c.is_whitespace() || matches!(c, '<' | '[' | ']' | '"' | '\'' | ')')).unwrap_or(tail.len());
let path = &tail[..len];
// Which domain, if any, is written in front of it.
let before = &body[..at];
let origin = origins.iter().find(|o| before.ends_with(o.as_str()));
let start = at - origin.map_or(0, |o| o.len());
if origin.is_none() {
// Bare: only standing on its own, as the short attachment form must.
let standing = matches!(before.chars().next_back(), None | Some(' ' | '\n' | '\t' | ']' | '(' | '>'));
if !standing {
continue;
}
} else if start < cursor {
continue;
}
let probe = format!("https://uploads.{KIWIFARMS_CLEARNET_HOST}{path}");
if !find_upload_url(&probe).is_some_and(|(matched, ..)| matched == probe) {
continue;
}
out.push_str(&body[cursor..start]);
out.push_str(&format!("https://{}{path}", upload_host(host)));
cursor = at + len;
search = cursor;
changed = true;
}
if !changed {
return std::borrow::Cow::Borrowed(body);
}
out.push_str(&body[cursor..]);
std::borrow::Cow::Owned(out)
}
fn expand_attachment_pages<'a>(body: &'a str, host: &str) -> std::borrow::Cow<'a, str> {
const MARKER: &str = "/attachments/";
if !body.contains(MARKER) {
return std::borrow::Cow::Borrowed(body);
@@ -440,12 +629,7 @@ pub(super) fn expand_attachment_paths<'a>(body: &'a str, host: &str) -> std::bor
/// the same message with a working embed a moment later instead of
/// stalling the whole room's message processing on one slow Tor fetch.
pub(super) fn spawn_attachment_resolve(state: AppState, http: http::HttpClient, host: &str, buffer_id: String, msg_id: String, body: String) {
let Some((matched, id, ext)) = find_attachment_url(&body) else { return };
let (matched, id, ext) = (matched.to_string(), id.to_string(), ext.to_ascii_lowercase());
let fetch_url = match matched.find("/attachments/") {
Some(at) => format!("https://{host}{}", &matched[at..]),
None => return,
};
let Some(Resolvable { matched, id, ext, fetch_url, page_url: page }) = find_resolvable(&body, host) else { return };
tokio::spawn(async move {
let dir = attachment_cache_dir();
@@ -464,15 +648,15 @@ pub(super) fn spawn_attachment_resolve(state: AppState, http: http::HttpClient,
}
}
Ok(Ok((status, _))) => {
tracing::debug!("sneedchat: attachment {id} fetch returned HTTP {status}");
tracing::info!("sneedchat: attachment {id} ({fetch_url}) answered HTTP {status}");
None
}
Ok(Err(e)) => {
tracing::debug!("sneedchat: fetching attachment {id}: {e}");
tracing::info!("sneedchat: fetching attachment {id}: {e}");
None
}
Err(_) => {
tracing::debug!("sneedchat: attachment {id} fetch timed out");
tracing::info!("sneedchat: attachment {id} fetch timed out");
None
}
}
@@ -484,7 +668,6 @@ pub(super) fn spawn_attachment_resolve(state: AppState, http: http::HttpClient,
// and the link it replaces leaves the text, as the old rewrite did.
if let Some(local_url) = local_url {
let kind = if matches!(ext.as_str(), "mp4" | "webm" | "mov" | "mkv") { "video" } else { "image" };
let page = matched.find("/attachments/").map(|at| format!("https://{KIWIFARMS_CLEARNET_HOST}{}", &matched[at..]));
let attachment = crate::model::Attachment {
kind: kind.to_string(),
filename: Some(format!("{id}.{ext}")),
@@ -492,8 +675,7 @@ pub(super) fn spawn_attachment_resolve(state: AppState, http: http::HttpClient,
url: page,
..Default::default()
};
let new_body = body.replace(&matched, "");
state.runtime.update_message_body_only(&state, &buffer_id, &msg_id, new_body.trim());
state.runtime.update_message_body_only(&state, &buffer_id, &msg_id, &without_link(&body, &matched));
let attachments = vec![attachment];
if let Err(e) = state.store.update_message_attachments(&buffer_id, &msg_id, &attachments) {
tracing::debug!("sneedchat: recording attachment {id}: {e}");
@@ -657,4 +839,122 @@ mod tests {
assert_eq!(cached_avatar_file(&dir, "8").await.as_deref(), Some(jpg.as_path()));
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn a_direct_picture_link_from_the_uploads_host_is_found() {
let body = "[img]https://uploads.kiwifarms.st/data/attachments/9671/9671535-a2e5ae0d55aadd338162770b3fa79f97-alt.avif?hash=-d_TarFo41[/img]";
let (matched, id, ext, ask) = find_upload_url(body).expect("a link");
assert_eq!(matched, "https://uploads.kiwifarms.st/data/attachments/9671/9671535-a2e5ae0d55aadd338162770b3fa79f97-alt.avif?hash=-d_TarFo41");
assert_eq!((id.as_str(), ext.as_str()), ("9671535", "avif"));
assert_eq!(ask, "/data/attachments/9671/9671535-a2e5ae0d55aadd338162770b3fa79f97-alt.avif?hash=-d_TarFo41");
}
#[test]
fn a_video_link_with_or_without_its_hash_is_found() {
let (matched, id, ext, ask) = find_upload_url("look https://uploads.kiwifarms.st/data/video/9667/9667943-6f0fa527b905c385fe7c60efee19011c.mp4?hash=pj-hdmJ9mQ so good").unwrap();
assert!(matched.ends_with("hash=pj-hdmJ9mQ"));
assert_eq!((id.as_str(), ext.as_str()), ("9667943", "mp4"));
assert!(ask.starts_with("/data/video/9667/"));
let (matched, ..) = find_upload_url("@skrod, https://uploads.kiwifarms.st/data/video/9663/9663232-d3e363a800d4828bc2fd29edbde6dfbd.mp4").unwrap();
assert!(matched.ends_with(".mp4"));
}
#[test]
fn nothing_but_the_forums_own_uploads_host_and_shape_is_taken() {
for not in [
"https://example.com/data/attachments/9671/9671535-abc-alt.avif",
"https://uploads.kiwifarms.st/data/attachments/9671/9671535-abc.exe",
"https://uploads.kiwifarms.st/data/other/9671/9671535-abc.avif",
"https://uploads.kiwifarms.st/data/attachments/x/9671535-abc.avif",
"https://uploads.kiwifarms.st/data/attachments/9671/notanumber-abc.avif",
"https://uploads.kiwifarms.st/data/attachments/9671/9671535.avif",
"https://uploads.kiwifarms.st/data/attachments/9671/../9671535-abc.avif",
] {
assert!(find_upload_url(not).is_none(), "{not}");
}
}
#[test]
fn the_earliest_of_the_two_kinds_is_the_one_fetched() {
let page = "https://kiwifarms.st/attachments/cat-png.123/";
let upload = "https://uploads.kiwifarms.st/data/attachments/9671/9671535-abc-alt.avif?hash=x";
let r = find_resolvable(&format!("{upload} then {page}"), "kiwifarms.st").unwrap();
assert_eq!(r.id, "9671535");
let r = find_resolvable(&format!("{page} then {upload}"), "kiwifarms.st").unwrap();
assert_eq!(r.id, "123");
}
#[test]
fn it_is_asked_for_where_this_account_reaches_the_forum() {
let upload = "https://uploads.kiwifarms.st/data/attachments/9671/9671535-abc-alt.avif?hash=x";
assert_eq!(find_resolvable(upload, "kiwifarms.st").unwrap().fetch_url, upload, "on the open internet, its uploads host");
let through_tor = find_resolvable(upload, DEFAULT_ONION).unwrap().fetch_url;
assert_eq!(through_tor, format!("https://{DEFAULT_ONION}/data/attachments/9671/9671535-abc-alt.avif?hash=x"), "through Tor, the onion address");
}
#[test]
fn the_tags_that_held_a_link_go_with_it() {
let link = "https://uploads.kiwifarms.st/data/attachments/9671/9671535-abc-alt.avif?hash=x";
assert_eq!(without_link(&format!("[img]{link}[/img]"), link), "");
assert_eq!(without_link(&format!("reminds me of him [img]{link}[/img]"), link), "reminds me of him");
assert_eq!(without_link(&format!("look {link}"), link), "look");
}
const UPLOAD: &str = "/data/attachments/9671/9671535-a2e5ae0d55aadd338162770b3fa79f97-alt.avif?hash=-d_TarFo41";
#[test]
fn a_direct_file_link_loses_its_domain_and_is_given_the_one_this_account_reaches_the_forum_at() {
// As the chat has it: wrapped, with the uploads host.
let theirs = format!("[img]https://uploads.kiwifarms.st{UPLOAD}[/img]");
assert_eq!(expand_attachment_paths(&theirs, KIWIFARMS_CLEARNET_HOST), theirs, "on the open internet it is already the right one");
// Through Tor, the onion address: the clearnet one is not asked.
assert_eq!(expand_attachment_paths(&theirs, DEFAULT_ONION), format!("[img]https://{DEFAULT_ONION}{UPLOAD}[/img]"));
}
#[test]
fn whichever_address_the_link_was_copied_with_it_comes_out_as_this_accounts_own() {
let want_clear = format!("https://uploads.kiwifarms.st{UPLOAD}");
let want_onion = format!("https://{DEFAULT_ONION}{UPLOAD}");
for origin in ["https://uploads.kiwifarms.st", "https://kiwifarms.st", "https://www.kiwifarms.st", &format!("https://{DEFAULT_ONION}"), &format!("http://{DEFAULT_ONION}")] {
let text = format!("look {origin}{UPLOAD} lol");
assert_eq!(expand_attachment_paths(&text, KIWIFARMS_CLEARNET_HOST), format!("look {want_clear} lol"), "{origin}");
assert_eq!(expand_attachment_paths(&text, DEFAULT_ONION), format!("look {want_onion} lol"), "{origin}");
}
}
#[test]
fn typed_short_it_is_the_same() {
assert_eq!(expand_attachment_paths(UPLOAD, KIWIFARMS_CLEARNET_HOST), format!("https://uploads.kiwifarms.st{UPLOAD}"));
assert_eq!(expand_attachment_paths(&format!("[img]{UPLOAD}[/img]"), DEFAULT_ONION), format!("[img]https://{DEFAULT_ONION}{UPLOAD}[/img]"));
assert_eq!(
expand_attachment_paths("/data/video/9667/9667943-6f0fa527b905c385fe7c60efee19011c.mp4?hash=pj-hdmJ9mQ", DEFAULT_ONION),
format!("https://{DEFAULT_ONION}/data/video/9667/9667943-6f0fa527b905c385fe7c60efee19011c.mp4?hash=pj-hdmJ9mQ")
);
}
#[test]
fn only_the_forums_own_files_on_its_own_addresses_are_touched() {
for left_alone in [
"https://example.com/data/attachments/9671/9671535-abc-alt.avif",
"https://kiwifarms.st/threads/some-thread.123/",
"https://kiwifarms.st/data/other/9671/9671535-abc.avif",
"see site/data/attachments/9671/9671535-abc-alt.avif",
"/data/attachments/9671/9671535-abc.exe",
"the /data/ folder",
] {
assert_eq!(expand_attachment_paths(left_alone, DEFAULT_ONION), left_alone, "{left_alone}");
}
}
#[test]
fn what_comes_out_is_something_the_resolver_fetches_through_the_account() {
let text = expand_attachment_paths(&format!("[img]https://uploads.kiwifarms.st{UPLOAD}[/img]"), DEFAULT_ONION).into_owned();
let r = find_resolvable(&text, DEFAULT_ONION).expect("fetchable");
assert_eq!(r.id, "9671535");
assert_eq!(r.fetch_url, format!("https://{DEFAULT_ONION}{UPLOAD}"));
// And both kinds of link in one message are each written for the account.
let both = expand_attachment_paths("/attachments/seal-png.9609686/ and https://uploads.kiwifarms.st/data/video/9667/9667943-abc.mp4", DEFAULT_ONION);
assert!(both.contains(&format!("https://{DEFAULT_ONION}/attachments/seal-png.9609686/")));
assert!(both.contains(&format!("https://{DEFAULT_ONION}/data/video/9667/9667943-abc.mp4")));
}
}
+316
View File
@@ -0,0 +1,316 @@
//! A link to a GIF, drawn as the GIF.
//!
//! Tenor, Giphy and Klipy are where nearly every GIF in a chat comes from, and
//! a link to one is a link to a page: nothing in its text says it is a moving
//! picture, so a line that was just such a link showed as an address.
//!
//! Discord settles it for its own messages - it unfurls the link and sends the
//! picture with the message (see discord::messages::gif_embed). Nothing else
//! does: an IRC line, a Sneedchat post, a Kick message are a bare address. So
//! here, for every service but Discord, the page is found for what it is.
//!
//! What can be done differs by site, and only what can be done honestly is:
//!
//! - **Tenor** says what its pages show in OpenGraph tags, one request for the
//! page and nothing more.
//! - **Giphy** puts the picture at an address made of the page's own id, so no
//! request is made at all.
//! - **Klipy** puts its pages behind a challenge that a program cannot pass, and
//! is left alone rather than got round. Its pictures arrive as such wherever a
//! Discord account carries them, and its direct links - which end in a file
//! extension - are drawn like any other.
//!
//! Like the YouTube cards (unfurl.rs) this is the account's own request: it goes
//! the account's way, through Tor where the account is, and is never made
//! directly instead when no route is ready.
use crate::model::Embed;
use crate::state::AppState;
use std::collections::HashMap;
use std::sync::{LazyLock, Mutex};
use std::time::Duration;
use tokio::sync::Semaphore;
/// A link to a GIF's page, and what it is on.
#[derive(Debug, PartialEq, Eq, Clone)]
pub enum GifLink {
Tenor { url: String },
Giphy { url: String, id: String },
}
impl GifLink {
fn url(&self) -> &str {
match self {
GifLink::Tenor { url } | GifLink::Giphy { url, .. } => url,
}
}
}
/// The address a link ends at: up to whitespace or the punctuation that closes
/// a sentence or a bracket round it.
fn address_at(text: &str, start: usize) -> String {
let rest = &text[start..];
let end = rest.find(|c: char| c.is_whitespace() || matches!(c, '<' | '>' | '[' | ']' | '"' | '\'')).unwrap_or(rest.len());
rest[..end].trim_end_matches(['.', ',', ';', ':', '!', '?', ')']).to_string()
}
/// Where a page of a given site is, anywhere in `text`: the earliest link of
/// either kind.
pub fn find_link(text: &str) -> Option<GifLink> {
let mut best: Option<(usize, GifLink)> = None;
let mut consider = |at: usize, link: GifLink| {
if best.as_ref().is_none_or(|(pos, _)| at < *pos) {
best = Some((at, link));
}
};
for scheme_host in ["https://tenor.com/", "https://www.tenor.com/", "http://tenor.com/"] {
if let Some(at) = text.find(scheme_host) {
let url = address_at(text, at);
// `/view/` after an optional locale: `/de/view/`, `/en-GB/view/`.
let path = &url[scheme_host.len()..];
let after_locale = match path.split_once('/') {
Some((first, rest)) if first.len() <= 5 && first != "view" && !rest.is_empty() => rest,
_ => path,
};
if after_locale.starts_with("view/") && after_locale.len() > "view/".len() {
consider(at, GifLink::Tenor { url });
}
}
}
for scheme_host in ["https://giphy.com/gifs/", "https://www.giphy.com/gifs/", "https://giphy.com/embed/"] {
if let Some(at) = text.find(scheme_host) {
let url = address_at(text, at);
let tail = url[scheme_host.len()..].split(['?', '#', '/']).next().unwrap_or("");
// `slug-with-hyphens-AbC123`: the id is what follows the last hyphen.
let id = tail.rsplit('-').next().unwrap_or("").to_string();
if id.len() >= 6 && id.chars().all(|c| c.is_ascii_alphanumeric()) {
consider(at, GifLink::Giphy { url, id });
}
}
}
best.map(|(_, link)| link)
}
/// What a `<meta>` tag with this `property` says, from a page's HTML.
fn meta_content(html: &str, property: &str) -> Option<String> {
let needle_double = format!("property=\"{property}\"");
let needle_single = format!("property='{property}'");
let mut from = 0;
while let Some(open) = html[from..].find("<meta") {
let start = from + open;
let end = html[start..].find('>').map_or(html.len(), |e| start + e);
let tag = &html[start..end];
if tag.contains(&needle_double) || tag.contains(&needle_single) {
let content = tag.find("content=\"").map(|c| (c + 9, '"')).or_else(|| tag.find("content='").map(|c| (c + 9, '\'')))?;
let value = &tag[content.0..];
let value = &value[..value.find(content.1)?];
return Some(value.replace("&amp;", "&"));
}
from = end;
}
None
}
/// A Tenor page's GIF, from the tags it describes itself with.
///
/// The video is preferred to the image: it is a fraction of the size, and the
/// window plays one as it plays any inline video, looped and silent.
fn from_tenor_page(html: &str, url: &str) -> Option<Embed> {
let video = meta_content(html, "og:video").or_else(|| meta_content(html, "og:video:secure_url"));
let kind = if video.is_some() { "video" } else { "image" };
let media = video.or_else(|| meta_content(html, "og:image"))?;
// Only from Tenor's own media hosts: a page cannot make this say "go and
// load that" about anywhere else.
if !is_tenor_media(&media) {
return None;
}
let size = |p: &str| meta_content(html, p).and_then(|v| v.parse::<u32>().ok()).filter(|&n| n > 0);
Some(Embed {
kind: Some("gif".to_string()),
url: Some(url.to_string()),
media_url: Some(media),
media_kind: Some(kind.to_string()),
width: size("og:image:width").or_else(|| size("og:video:width")),
height: size("og:image:height").or_else(|| size("og:video:height")),
provider: Some("Tenor".to_string()),
..Default::default()
})
}
fn is_tenor_media(url: &str) -> bool {
let host = url.strip_prefix("https://").and_then(|rest| rest.split('/').next()).unwrap_or("");
host == "tenor.com" || host.ends_with(".tenor.com")
}
/// A Giphy GIF, from its id alone.
fn from_giphy_id(url: &str, id: &str) -> Embed {
Embed {
kind: Some("gif".to_string()),
url: Some(url.to_string()),
media_url: Some(format!("https://media.giphy.com/media/{id}/giphy.mp4")),
media_kind: Some("video".to_string()),
provider: Some("Giphy".to_string()),
..Default::default()
}
}
/// Pages already looked at, and ones that said nothing. Bounded crudely: all it
/// saves is a repeat request.
static KNOWN: LazyLock<Mutex<HashMap<String, Option<Embed>>>> = LazyLock::new(|| Mutex::new(HashMap::new()));
const KNOWN_LIMIT: usize = 512;
/// A backlog full of links is read two at a time, not all at once.
static SLOTS: Semaphore = Semaphore::const_new(2);
const TIMEOUT: Duration = Duration::from_secs(15);
/// Puts the GIF on the message if its text has a link to one. Spawned: the
/// message is already out, and nothing here holds it up.
///
/// Not for Discord, which has already said what its links are.
pub fn gif_later(state: &AppState, account_id: &str, buffer_id: &str, msg_id: &str, body: &str) {
if account_id.starts_with("discord:") {
return;
}
let Some(link) = find_link(body) else { return };
let (state, account_id, buffer_id, msg_id) =
(state.clone(), account_id.to_string(), buffer_id.to_string(), msg_id.to_string());
tokio::spawn(async move {
let Some(embed) = describe(&state, &account_id, &link).await else { return };
crate::unfurl::add_embed(&state, &buffer_id, &msg_id, embed);
});
}
async fn describe(state: &AppState, account_id: &str, link: &GifLink) -> Option<Embed> {
// Giphy needs nothing from anybody.
if let GifLink::Giphy { url, id } = link {
return Some(from_giphy_id(url, id));
}
if let Some(known) = KNOWN.lock().unwrap().get(link.url()) {
return known.clone();
}
let _slot = SLOTS.acquire().await.ok()?;
if let Some(known) = KNOWN.lock().unwrap().get(link.url()) {
return known.clone();
}
let router = crate::net::route::router();
let routed = router.tunnel_all()
|| state.accounts.route_level_of(account_id).is_some_and(|level| level != "clearnet");
if routed && router.ready(|_| {}).await.is_err() {
// No route. Not remembered: the next message may find one.
return None;
}
let client = router.client_if("unfurl", routed, |builder| builder.timeout(TIMEOUT).user_agent(concat!("moho/", env!("CARGO_PKG_VERSION"))));
let response = client.get(link.url()).send().await.ok()?;
let status = response.status();
if !status.is_success() {
// A page that is gone is gone; a refusal says nothing about the GIF.
if matches!(status.as_u16(), 400 | 404 | 410) {
remember(link.url(), None);
}
return None;
}
// Only the head of the page: the tags are in it, and the rest is not worth
// reading however long it is.
let html = response.text().await.ok()?;
let head = &html[..html.len().min(256 * 1024)];
let found = match link {
GifLink::Tenor { url } => from_tenor_page(head, url),
GifLink::Giphy { .. } => None,
};
remember(link.url(), found.clone());
found
}
fn remember(url: &str, embed: Option<Embed>) {
let mut known = KNOWN.lock().unwrap();
if known.len() >= KNOWN_LIMIT {
known.clear();
}
known.insert(url.to_string(), embed);
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn finds_a_tenor_page_and_where_it_ends() {
let text = "look at this https://tenor.com/view/rickroll-roll-rick-never-gonna-give-you-up-gif-22954713, so good";
assert_eq!(
find_link(text),
Some(GifLink::Tenor { url: "https://tenor.com/view/rickroll-roll-rick-never-gonna-give-you-up-gif-22954713".to_string() })
);
}
#[test]
fn a_tenor_page_may_carry_a_locale() {
assert!(matches!(find_link("https://tenor.com/de/view/x-gif-1"), Some(GifLink::Tenor { .. })));
assert!(matches!(find_link("https://tenor.com/en-GB/view/x-gif-1"), Some(GifLink::Tenor { .. })));
}
#[test]
fn the_rest_of_tenor_is_not_a_gif() {
assert_eq!(find_link("https://tenor.com/"), None);
assert_eq!(find_link("https://tenor.com/search/cats-gifs"), None);
assert_eq!(find_link("https://tenor.com/view/"), None);
}
#[test]
fn a_giphy_page_is_known_by_its_id_alone() {
let link = find_link("https://giphy.com/gifs/cat-vibing-music-Z7wFi5b2VEAr6").unwrap();
assert_eq!(link, GifLink::Giphy { url: "https://giphy.com/gifs/cat-vibing-music-Z7wFi5b2VEAr6".to_string(), id: "Z7wFi5b2VEAr6".to_string() });
let embed = from_giphy_id("https://giphy.com/gifs/x-Z7wFi5b2VEAr6", "Z7wFi5b2VEAr6");
assert_eq!(embed.media_url.as_deref(), Some("https://media.giphy.com/media/Z7wFi5b2VEAr6/giphy.mp4"));
assert_eq!(embed.kind.as_deref(), Some("gif"));
// A bare id, with no slug.
assert!(matches!(find_link("https://giphy.com/gifs/Z7wFi5b2VEAr6"), Some(GifLink::Giphy { .. })));
assert_eq!(find_link("https://giphy.com/gifs/"), None);
}
#[test]
fn the_earliest_link_wins() {
let text = "https://giphy.com/gifs/a-Z7wFi5b2VEAr6 and https://tenor.com/view/b-gif-2";
assert!(matches!(find_link(text), Some(GifLink::Giphy { .. })));
let text = "https://tenor.com/view/b-gif-2 and https://giphy.com/gifs/a-Z7wFi5b2VEAr6";
assert!(matches!(find_link(text), Some(GifLink::Tenor { .. })));
}
/// The tags a Tenor page describes itself with, as it sends them.
const PAGE: &str = r#"<html><head>
<meta class="dynamic" property="og:image" content="https://media1.tenor.com/m/x8v1oNUOmg4AAAAd/rickroll-roll.gif">
<meta class="dynamic" property="og:image:width" content="640">
<meta class="dynamic" property="og:image:height" content="480">
<meta class="dynamic" property="og:video" content="https://media.tenor.com/x8v1oNUOmg4AAAPo/rickroll-roll.mp4?a=1&amp;b=2">
</head></html>"#;
#[test]
fn a_tenor_page_gives_its_video_and_its_size() {
let e = from_tenor_page(PAGE, "https://tenor.com/view/x-gif-1").unwrap();
assert_eq!(e.media_url.as_deref(), Some("https://media.tenor.com/x8v1oNUOmg4AAAPo/rickroll-roll.mp4?a=1&b=2"));
assert_eq!((e.width, e.height), (Some(640), Some(480)));
assert_eq!(e.provider.as_deref(), Some("Tenor"));
assert_eq!(e.media_kind.as_deref(), Some("video"));
}
#[test]
fn with_no_video_the_picture_stands_in() {
let page = r#"<meta property="og:image" content="https://media1.tenor.com/m/a/b.gif">"#;
let e = from_tenor_page(page, "https://tenor.com/view/x-gif-1").unwrap();
assert_eq!(e.media_url.as_deref(), Some("https://media1.tenor.com/m/a/b.gif"));
assert_eq!(e.media_kind.as_deref(), Some("image"));
}
#[test]
fn a_page_cannot_point_somewhere_that_is_not_tenor() {
let page = r#"<meta property="og:video" content="https://evil.example/x.mp4">"#;
assert!(from_tenor_page(page, "https://tenor.com/view/x-gif-1").is_none());
let page = r#"<meta property="og:video" content="https://tenor.com.evil.example/x.mp4">"#;
assert!(from_tenor_page(page, "https://tenor.com/view/x-gif-1").is_none());
assert!(from_tenor_page("<html></html>", "https://tenor.com/view/x-gif-1").is_none());
}
}
+1
View File
@@ -5,6 +5,7 @@ mod backend;
mod commands;
mod events;
mod export;
mod gifs;
mod highlights;
mod ignores;
mod ipc;
+19
View File
@@ -91,6 +91,11 @@ pub struct Account {
/// nowhere to put one.
#[serde(rename = "statusText", default)]
pub status_text: String,
/// Shown as idle because the person has stepped away, and not because they
/// said so: `status` then reads idle while what they chose is online, and
/// coming back puts it right without anyone touching it.
#[serde(rename = "autoIdle", default, skip_serializing_if = "std::ops::Not::not")]
pub auto_idle: bool,
#[serde(rename = "displayName")]
pub display_name: String,
pub state: String,
@@ -530,6 +535,20 @@ pub struct Embed {
/// A glyph for the card's heading: "warning" for the ones that act on an account.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub icon: Option<String>,
/// Where the moving picture itself is, for `kind` "gif": a video or an image
/// a window can play as it stands, which is not the page the link led to.
#[serde(rename = "mediaUrl", default, skip_serializing_if = "Option::is_none")]
pub media_url: Option<String>,
/// "video" or "image": which of the two `media_url` is, said outright because
/// the address often cannot say - a proxied one ends in whatever the original
/// did, and a guess from the ending is wrong for the ones that have none.
#[serde(rename = "mediaKind", default, skip_serializing_if = "Option::is_none")]
pub media_kind: Option<String>,
/// Its size, so the row can be laid out before a byte of it has arrived.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub width: Option<u32>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub height: Option<u32>,
}
/// A link with its words.
+72 -1
View File
@@ -341,7 +341,10 @@ pub async fn dispatch(
if after > 0 {
let limit = p_i64(params, "limit", 200);
return match state.store.messages_after(buffer_id, after, limit) {
Ok(rows) => (Some(serde_json::to_value(rows).unwrap_or_default()), None),
Ok(rows) => {
crate::unfurl::backlog_later(state, buffer_id, &rows);
(Some(serde_json::to_value(rows).unwrap_or_default()), None)
}
Err(e) => (None, Some(format!("{e:#}"))),
};
}
@@ -444,6 +447,8 @@ pub async fn dispatch(
// previews are already showing, and re-signed links
// arrive as messageUpdated.
backend::discord::resign_stale_attachments(state.clone(), buffer_id.to_string(), &messages);
// And any link in it that was stored without a card.
crate::unfurl::backlog_later(state, buffer_id, &messages);
(Some(serde_json::to_value(messages).unwrap()), None)
}
Err(e) => (None, Some(format!("{e:#}"))),
@@ -969,6 +974,25 @@ pub async fn dispatch(
}
}
// The same for a Discord channel or direct message: set on the
// account, so it is quiet on a phone as well.
"setDiscordChannelMuted" => {
let Some(buffer_id) = p_str_opt(params, "bufferId") else {
return (None, Some("setDiscordChannelMuted requires \"bufferId\"".to_string()));
};
let muted = params.get("muted").and_then(|v| v.as_bool()).unwrap_or(true);
let Some(buffer) = state.runtime.get_buffer(buffer_id) else {
return (None, Some("no such conversation".to_string()));
};
if !buffer.account_id.starts_with("discord:") {
return (Some(ok_node()), None);
}
match backend::discord::mutes::set_muted(state, &buffer.account_id, buffer_id, muted).await {
Ok(()) => (Some(ok_node()), None),
Err(e) => (None, Some(format!("{e:#}"))),
}
}
"listMentionRoles" => {
let Some(buffer_id) = p_str_opt(params, "bufferId") else {
return (None, Some("listMentionRoles requires \"bufferId\"".to_string()));
@@ -1522,6 +1546,53 @@ pub async fn dispatch(
(Some(serde_json::json!({ "ok": true, "applied": applied })), None)
}
// GIFs to browse: what is trending, or what matches a search. Found
// through a Discord account, and sent anywhere.
"listGifs" => {
let query = p_str_opt(params, "query").unwrap_or("");
match backend::discord::list_gifs(state, query).await {
Ok(answer) => (Some(answer), None),
Err(e) => (None, Some(format!("{e:#}"))),
}
}
// The person has stepped away from the machine, or come back.
//
// Said once by whoever watches for it - the window, which can see how
// long it has been since anything was touched - and applied here, where
// the accounts and what each was set to are known. Only an account that
// is online is moved to idle and back; anything somebody chose is theirs
// until they change it (see Runtime::auto_idle_applies).
"setUserIdle" => {
let idle = params.get("idle").and_then(|v| v.as_bool()).unwrap_or(false);
if !state.runtime.set_user_idle(idle) {
return (Some(ok_node()), None);
}
let mut changed = Vec::new();
for account in state.runtime.list_accounts(state) {
let id = account.id.as_str();
if !crate::model::status_supported(id, "idle") || state.runtime.account_status(id) != "online" {
continue;
}
let shown = state.runtime.effective_status(state, id);
let applied = if state.accounts.get_discord(id).is_some() {
backend::discord::apply_session_presence(state, id, &shown)
} else if let Some(sender) = state.runtime.irc_sender(id) {
let away = if shown == "idle" { Some("Idle".to_string()) } else { None };
sender.send(irc::proto::Command::AWAY(away)).is_ok()
} else {
false
};
if applied {
changed.push(serde_json::json!({ "accountId": id, "status": shown, "autoIdle": shown == "idle" }));
}
}
if !changed.is_empty() {
state.events.emit("accountStatus", serde_json::json!({ "accounts": changed }));
}
(Some(ok_node()), None)
}
// Asks for a channel's member list.
//
// Separate from subscribe because a client subscribes to every buffer
+108 -1
View File
@@ -629,6 +629,9 @@ pub struct Runtime {
account_status: Mutex<HashMap<String, String>>,
/// What each account says beside its status, where the service has a place for it.
account_status_text: Mutex<HashMap<String, String>>,
/// Whether the person has stepped away from the machine, as the window last
/// said: what turns an account that is online into one that is idle.
user_idle: std::sync::atomic::AtomicBool,
discord_member_list_targets: Mutex<HashMap<(String, String), String>>,
/// (account, guild) -> that guild's voice channels, as (id, name, limit).
discord_voice_channels: Mutex<HashMap<(String, String), Vec<VoiceChannelEntry>>>,
@@ -1095,6 +1098,7 @@ impl Runtime {
matrix_space_parents: Mutex::new(HashMap::new()),
account_status: Mutex::new(HashMap::new()),
account_status_text: Mutex::new(HashMap::new()),
user_idle: std::sync::atomic::AtomicBool::new(false),
discord_member_list_targets: Mutex::new(HashMap::new()),
discord_voice_channels: Mutex::new(HashMap::new()),
discord_voice_states: Mutex::new(HashMap::new()),
@@ -1157,7 +1161,8 @@ impl Runtime {
// The constructors have no view of live state, so the status the
// runtime is actually holding is filled in here.
for account in &mut out {
account.status = self.account_status(&account.id);
account.auto_idle = self.auto_idle_applies(state, &account.id);
account.status = self.effective_status(state, &account.id);
account.status_text = self.account_status_text(&account.id);
// Who the service knows this account as, which is not the same
// question as what it is called here: a local rename moves
@@ -2820,6 +2825,42 @@ impl Runtime {
self.account_status_text.lock().unwrap().get(account_id).cloned().unwrap_or_default()
}
/// Records whether the person is away, and whether that changed.
pub fn set_user_idle(&self, idle: bool) -> bool {
self.user_idle.swap(idle, std::sync::atomic::Ordering::SeqCst) != idle
}
pub fn user_idle(&self) -> bool {
self.user_idle.load(std::sync::atomic::Ordering::SeqCst)
}
/// Whether this account is idle because its person is away, rather than
/// because they chose it.
///
/// Only an account that is online is ever moved: Idle, Do not disturb and
/// Invisible are what somebody said, and are kept until they say otherwise.
/// Only where the service has an idle to move to, and not while in a call,
/// which is somebody here by definition - the same two rules Discord's own
/// client keeps.
pub fn auto_idle_applies(&self, state: &AppState, account_id: &str) -> bool {
auto_idle_rule(
self.user_idle(),
&self.account_status(account_id),
crate::model::status_supported(account_id, "idle"),
state.voice.current_channel(account_id).is_some(),
)
}
/// What the account shows as: what its person chose, unless they are away
/// and it is online.
pub fn effective_status(&self, state: &AppState, account_id: &str) -> String {
if self.auto_idle_applies(state, account_id) {
"idle".to_string()
} else {
self.account_status(account_id)
}
}
pub fn set_account_status(&self, account_id: &str, status: &str) {
self.account_status.lock().unwrap().insert(account_id.to_string(), status.to_string());
}
@@ -4478,6 +4519,8 @@ impl Runtime {
// holding it up.
if message.embeds.is_empty() && kind != "system" {
crate::unfurl::youtube_later(state, account_id, &message.buffer_id, &message.id, &message.body);
// And a link to a GIF, where nothing has said what it is.
crate::gifs::gif_later(state, account_id, &message.buffer_id, &message.id, &message.body);
}
// Notify on every inbound DM regardless of content, or on a
@@ -4508,9 +4551,36 @@ impl Runtime {
/// never actually stored, e.g. seeing a message for the first time
/// that already carries prior edit history.
pub fn update_message(&self, state: &AppState, buffer_id: &str, msg_id: &str, body: &str, embeds: &[Embed], attachments: &[Attachment]) -> bool {
// An edit from a service that describes no links itself says nothing
// about cards, and is not a reason to lose them: the ones the daemon
// put on the message stay while their link is still in the text.
let describes_nothing = embeds.is_empty();
let embeds: Vec<Embed> = if describes_nothing {
state
.store
.get_message(buffer_id, msg_id)
.ok()
.flatten()
.map(|m| m.embeds)
.unwrap_or_default()
.into_iter()
.filter(|e| e.url.as_deref().is_some_and(|u| body.contains(u) || crate::unfurl::youtube_id(u).is_some_and(|id| crate::unfurl::youtube_id(body).as_deref() == Some(id.as_str()))))
.collect()
} else {
embeds.to_vec()
};
let embeds = &embeds[..];
match state.store.update_message_body(buffer_id, msg_id, body, embeds, attachments) {
Ok(true) => {
state.events.emit("messageUpdated", json!({ "bufferId": buffer_id, "id": msg_id, "body": body, "edited": true, "editedTs": chrono::Utc::now().timestamp(), "embeds": embeds, "attachments": attachments }));
// A link the edit added is described like any other; one
// already described is not asked about again.
if describes_nothing {
if let Some(buffer) = self.get_buffer(buffer_id) {
crate::unfurl::youtube_later(state, &buffer.account_id, buffer_id, msg_id, body);
crate::gifs::gif_later(state, &buffer.account_id, buffer_id, msg_id, body);
}
}
true
}
Ok(false) => false,
@@ -5218,6 +5288,43 @@ mod rename_tests {
}
}
/// Whether an account should show as idle because its person is away.
///
/// Only one that is online is moved: what somebody chose - idle, do not
/// disturb, invisible - stays until they choose again. Only where the service
/// has an idle, and not in a call.
pub(crate) fn auto_idle_rule(away: bool, chosen: &str, service_has_idle: bool, in_call: bool) -> bool {
away && chosen == "online" && service_has_idle && !in_call
}
#[cfg(test)]
mod auto_idle_tests {
use super::auto_idle_rule;
#[test]
fn an_online_account_goes_idle_when_its_person_is_away() {
assert!(auto_idle_rule(true, "online", true, false));
assert!(!auto_idle_rule(false, "online", true, false));
}
#[test]
fn what_somebody_chose_is_kept() {
for chosen in ["idle", "dnd", "invisible"] {
assert!(!auto_idle_rule(true, chosen, true, false), "{chosen}");
}
}
#[test]
fn a_service_with_no_idle_is_left_alone() {
assert!(!auto_idle_rule(true, "online", false, false));
}
#[test]
fn nobody_is_idle_in_a_call() {
assert!(!auto_idle_rule(true, "online", true, true));
}
}
/// Whether a key belongs to an account being forgotten.
///
/// One rule for every map in this struct, which is what makes forgetting an
+159 -11
View File
@@ -95,10 +95,63 @@ pub fn youtube_later(state: &AppState, account_id: &str, buffer_id: &str, msg_id
(state.clone(), account_id.to_string(), buffer_id.to_string(), msg_id.to_string());
tokio::spawn(async move {
let Some(embed) = describe(&state, &account_id, &id).await else { return };
state.runtime.set_message_embeds(&state, &buffer_id, &msg_id, &[embed]);
add_embed(&state, &buffer_id, &msg_id, embed);
});
}
/// Messages already asked about, and when. A conversation is read many times;
/// a link that could not be described should be tried again, but not on every
/// glance.
static ASKED: LazyLock<Mutex<HashMap<String, std::time::Instant>>> = LazyLock::new(|| Mutex::new(HashMap::new()));
const ASK_AGAIN_AFTER: Duration = Duration::from_secs(600);
/// How many of a page's messages are described at once, newest first.
const BACKLOG_LIMIT: usize = 30;
/// Describes the links in messages that were stored without a card.
///
/// Live messages are described as they arrive (`Runtime::record_message_at`),
/// but not everything reaches the store that way: Matrix history read back
/// from the server, Discord's catch-up, scrollback from before cards existed,
/// and a link that could not be described when it was said. Reading a
/// conversation is the moment to make up for those, whichever service it is
/// on.
pub fn backlog_later(state: &AppState, buffer_id: &str, messages: &[crate::model::Message]) {
let Some(account_id) = state.runtime.get_buffer(buffer_id).map(|b| b.account_id) else { return };
let now = std::time::Instant::now();
let mut asked = ASKED.lock().unwrap();
if asked.len() > 4096 {
asked.clear();
}
for message in messages.iter().rev().filter(|m| {
m.embeds.is_empty()
&& m.kind != "system"
&& (youtube_id(&m.body).is_some() || crate::gifs::find_link(&m.body).is_some())
}).take(BACKLOG_LIMIT) {
let key = format!("{buffer_id}\u{0}{}", message.id);
if asked.get(&key).is_some_and(|at| now.duration_since(*at) < ASK_AGAIN_AFTER) {
continue;
}
asked.insert(key, now);
youtube_later(state, &account_id, buffer_id, &message.id, &message.body);
crate::gifs::gif_later(state, &account_id, buffer_id, &message.id, &message.body);
}
}
/// Puts a card on a message beside whatever it already has.
///
/// More than one thing can describe a line - a YouTube link and a GIF in the
/// same message, each worked out on its own and finishing in either order - so
/// a card is added to what is there, not written over it. One that a message
/// already carries of the same link is not added twice.
pub fn add_embed(state: &AppState, buffer_id: &str, msg_id: &str, embed: Embed) {
let mut embeds = state.store.get_message(buffer_id, msg_id).ok().flatten().map(|m| m.embeds).unwrap_or_default();
if embeds.iter().any(|e| e.url.is_some() && e.url == embed.url) {
return;
}
embeds.push(embed);
state.runtime.set_message_embeds(state, buffer_id, msg_id, &embeds);
}
async fn describe(state: &AppState, account_id: &str, id: &str) -> Option<Embed> {
if let Some(known) = KNOWN.lock().unwrap().get(id) {
return known.clone();
@@ -127,21 +180,32 @@ async fn describe(state: &AppState, account_id: &str, id: &str) -> Option<Embed>
// Asked up to three times. Over Tor - Sneedchat's route - YouTube
// regularly turns an exit node away with a 403 or 429 that says nothing
// about the video, and a later try leaves by another one.
//
// And when oEmbed is refused outright, the video's own page is read for
// the same two facts. YouTube answers oEmbed with a 403 to every Tor exit,
// whichever circuit asks, while the page itself loads - so for an account
// on Tor the page is the only way a card is ever made.
let mut found = None;
for attempt in 0..3 {
if attempt > 0 {
tokio::time::sleep(Duration::from_secs(3 * attempt)).await;
}
let Ok(response) = client.get(&asked).send().await else { continue };
let status = response.status();
if status.is_success() {
let Ok(answer) = response.json::<serde_json::Value>().await else { continue };
found = Some(parse(&answer, &watch));
break;
} else if answers_for_the_video(status.as_u16()) {
// Private, removed, or not embeddable: YouTube's answer, and it
// will be the same next time.
found = Some(None);
if let Ok(response) = client.get(&asked).send().await {
let status = response.status();
if status.is_success() {
if let Ok(answer) = response.json::<serde_json::Value>().await {
found = Some(parse(&answer, &watch));
break;
}
} else if answers_for_the_video(status.as_u16()) {
// Private, removed, or not embeddable: YouTube's answer, and it
// will be the same next time.
found = Some(None);
break;
}
}
if let Some(card) = read_page(&client, &watch).await {
found = Some(Some(card));
break;
}
}
@@ -177,6 +241,73 @@ fn parse(answer: &serde_json::Value, watch: &str) -> Option<Embed> {
})
}
/// The card from the video's own page, for when oEmbed will not answer.
///
/// The page is a JavaScript application that carries its data inline, and
/// the title and channel are in it as JSON. Nothing promises where, so this
/// looks for the renderers that have held them and takes the first that is
/// there; a page that has none of them - a consent wall, a removed video, a
/// layout YouTube has since changed - makes no card, the same as no answer.
async fn read_page(client: &reqwest::Client, watch: &str) -> Option<Embed> {
let response = client
.get(watch)
.header("User-Agent", "Mozilla/5.0 (X11; Linux x86_64; rv:128.0) Gecko/20100101 Firefox/128.0")
.header("Accept-Language", "en-US,en;q=0.9")
// Declines the consent page an exit in Europe is otherwise sent to.
.header("Cookie", "SOCS=CAI")
.send()
.await
.ok()?;
if !response.status().is_success() {
return None;
}
parse_page(&response.text().await.ok()?, watch)
}
/// The JSON string that follows `marker`, decoded.
fn string_after(page: &str, marker: &str) -> Option<String> {
let rest = &page[page.find(marker)? + marker.len()..];
let mut end = None;
let mut escaped = false;
for (i, c) in rest.char_indices() {
match c {
_ if escaped => escaped = false,
'\\' => escaped = true,
'"' => {
end = Some(i);
break;
}
_ => {}
}
}
let literal = format!("\"{}\"", &rest[..end?]);
serde_json::from_str::<String>(&literal).ok().map(|v| v.trim().to_string()).filter(|v| !v.is_empty())
}
/// The card from a watch page's text. None without a title.
fn parse_page(page: &str, watch: &str) -> Option<Embed> {
let title = [
"\"videoDescriptionHeaderRenderer\":{\"title\":{\"runs\":[{\"text\":\"",
"\"videoPrimaryInfoRenderer\":{\"title\":{\"runs\":[{\"text\":\"",
"\"playerOverlayVideoDetailsRenderer\":{\"title\":{\"simpleText\":\"",
]
.iter()
.find_map(|marker| string_after(page, marker))?;
// The channel follows the title in the same renderer.
let author = page
.find("\"videoDescriptionHeaderRenderer\":{")
.and_then(|at| string_after(&page[at..], "\"channel\":{\"simpleText\":\""))
.or_else(|| string_after(page, "\"ownerChannelName\":\""));
Some(Embed {
title: Some(title),
url: Some(watch.to_string()),
provider: Some("YouTube".to_string()),
author,
color: Some(0xFF0000),
..Default::default()
})
}
#[cfg(test)]
mod tests {
use super::*;
@@ -236,4 +367,21 @@ mod tests {
assert_eq!(card.url.as_deref(), Some(watch));
assert!(parse(&serde_json::json!({ "title": " ", "author_name": "x" }), watch).is_none());
}
#[test]
fn a_card_from_the_page_when_oembed_is_refused() {
let watch = "https://www.youtube.com/watch?v=_-agl0pOQfs";
let page = r#"<script>var x={"contents":{"videoDescriptionHeaderRenderer":{"title":{"runs":[{"text":"Insane Clown Posse - \"Miracles\" (Official Music Video)"}]},"channel":{"simpleText":"Psychopathic Records"},"views":{"simpleText":"19,675,62"}}}}</script>"#;
let card = parse_page(page, watch).unwrap();
assert_eq!(card.title.as_deref(), Some("Insane Clown Posse - \"Miracles\" (Official Music Video)"));
assert_eq!(card.author.as_deref(), Some("Psychopathic Records"));
assert_eq!(card.url.as_deref(), Some(watch));
assert_eq!(card.provider.as_deref(), Some("YouTube"));
// The older layout, which names no channel in the same place.
let page = r#"{"videoPrimaryInfoRenderer":{"title":{"runs":[{"text":"A \u0026 B"}]},"viewCount":{}}}"#;
let card = parse_page(page, watch).unwrap();
assert_eq!(card.title.as_deref(), Some("A & B"));
assert_eq!(card.author, None);
// A consent wall, or a removed video, has no title: no card.
assert!(parse_page("<html><title> - YouTube</title></html>", watch).is_none());
}
}