Skip to main content

stygian_charon/
content_type_shift.rs

1//! Content-Type shift detection (T104) — rolling baseline tracker.
2//!
3//! The 2026 scraping guide
4//! (`docs/dev/project/scraping-guide-2026-llm-context.md` §"POST-EXTRACTION",
5//! L2536) calls out that publishers are starting to serve a deliberately
6//! different document to AI-bot User-Agents (HTML replaced with a
7//! Markdown stub, the same URL, the same `200`, the same parse
8//! succeeding). Selector-based validators that only count fields don't
9//! notice because the Markdown stub still has plenty of fields — they're
10//! just different fields.
11//!
12//! This module catches the publisher-cloaking pattern by tracking
13//! `(Content-Type, byte_length)` per identity and emitting a
14//! [`ContentTypeShiftReport`] when either dimension drifts past the
15//! configured threshold.
16
17use std::collections::VecDeque;
18use std::time::{SystemTime, UNIX_EPOCH};
19
20use serde::{Deserialize, Serialize};
21use thiserror::Error;
22
23/// Coarse MIME classification. Two responses count as "same class" only
24/// if they share the same variant — the `String` form (e.g.
25/// `text/html; charset=utf-8`) is normalised to the variant.
26#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
27pub enum MimeClass {
28    /// HTML / XHTML documents.
29    Html,
30    /// JSON documents (including vendor `+json` suffixes).
31    Json,
32    /// XML documents (including vendor `+xml` suffixes).
33    Xml,
34    /// Markdown documents — the canonical publisher-cloaking target.
35    Markdown,
36    /// Plain text documents.
37    Text,
38    /// Binary payloads (images, audio, video, `application/octet-stream`,
39    /// `application/pdf`).
40    Binary,
41    /// Anything else — empty, malformed, or a content type the
42    /// classifier doesn't recognise.
43    Unknown,
44}
45
46impl MimeClass {
47    /// Classify a `Content-Type` header value (case-insensitive,
48    /// parameters ignored).
49    #[must_use]
50    pub fn from_content_type(value: &str) -> Self {
51        let head = value
52            .split(';')
53            .next()
54            .unwrap_or("")
55            .trim()
56            .to_ascii_lowercase();
57        let Some((family, subtype)) = head.split_once('/') else {
58            return Self::Unknown;
59        };
60        match (family, subtype) {
61            ("text", s) if s == "html" || s == "xhtml" => Self::Html,
62            ("text", "markdown") => Self::Markdown,
63            // Both `application` and `text` families can carry JSON
64            // (including suffixes like `+json`). Match on the
65            // subtype that's already been narrowed by the guard.
66            _ if subtype == "json" || subtype.ends_with("+json") => Self::Json,
67            ("application", s) if s == "xml" || s == "xhtml" || s.ends_with("+xml") => Self::Xml,
68            ("text", _) => Self::Text,
69            ("image" | "audio" | "video", _) => Self::Binary,
70            ("application", s) if s == "octet-stream" || s == "pdf" => Self::Binary,
71            _ => Self::Unknown,
72        }
73    }
74}
75
76/// One observation per scrape.
77#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
78pub struct ContentTypeObservation {
79    /// Identity the observation is keyed under. Same identity → same
80    /// history slot.
81    pub key: String,
82    /// Raw `Content-Type` header value.
83    pub content_type: String,
84    /// Response body byte length.
85    pub byte_length: u64,
86}
87
88/// Detected drift between the most recent observation and the
89/// historical baseline.
90#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
91pub enum ContentTypeDrift {
92    /// `MimeClass` flipped from one variant to another. A
93    /// `Html → Markdown` shift is the canonical publisher-cloaking
94    /// signature.
95    ClassChange {
96        /// `MimeClass` of the previous observation (the baseline).
97        from: MimeClass,
98        /// `MimeClass` of the latest observation.
99        to: MimeClass,
100    },
101    /// Response body shrank by more than the configured threshold
102    /// (default 50 %). The Markdown-stub pattern shrinks bodies by
103    /// 20–100×.
104    ByteCollapse {
105        /// `byte_count_latest / byte_count_baseline`. Always in
106        /// `(0.0, 1.0]` for a true collapse.
107        ratio: f64,
108    },
109}
110
111/// Aggregated shift report for one identity.
112#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
113pub struct ContentTypeShiftReport {
114    /// Identity the report covers.
115    pub key: String,
116    /// The latest observation that triggered the report.
117    pub current: ContentTypeObservation,
118    /// Detected drift, if any.
119    pub drift: Option<ContentTypeDrift>,
120}
121
122/// Errors raised by [`ContentTypeShiftDetector::record`].
123#[derive(Debug, Clone, Error, PartialEq, Eq)]
124pub enum ContentTypeError {
125    /// Identity (`key`) is empty. Callers must supply a non-empty
126    /// stable identity — typically `(url, vendor_class, ua_fingerprint)`.
127    #[error("identity key is empty")]
128    EmptyKey,
129}
130
131/// Port trait for content-type shift detection.
132pub trait ContentTypeShiftDetector: Send + Sync {
133    /// Record one observation and return the shift report.
134    ///
135    /// # Errors
136    ///
137    /// Returns [`ContentTypeError::EmptyKey`] when the observation's
138    /// `key` is empty.
139    fn record(
140        &mut self,
141        observation: &ContentTypeObservation,
142    ) -> Result<ContentTypeShiftReport, ContentTypeError>;
143}
144
145/// Default rolling-baseline adapter.
146///
147/// Keeps the last 32 observations per identity (configurable via
148/// [`RollingBaselineDetector::with_capacity`]) and emits a drift
149/// report when the latest observation disagrees with the most
150/// recent baseline on either dimension.
151///
152/// `Send + Sync` is implemented by mutating via [`std::sync::Mutex`] —
153/// the detector is shared across snapshot runs in the existing
154/// `SnapshotBuilder` (T88) which already passes the detector by
155/// `Arc`.
156#[derive(Debug)]
157pub struct RollingBaselineDetector {
158    /// Per-identity ring buffer of observations (oldest first).
159    history: std::sync::Mutex<std::collections::HashMap<String, VecDeque<ContentTypeObservation>>>,
160    /// Max observations retained per identity (ring-buffer cap).
161    capacity: usize,
162    /// `byte_count_latest / byte_count_baseline` threshold below which
163    /// a `ByteCollapse` is emitted. Default 0.5 (50 %).
164    byte_collapse_threshold: f64,
165}
166
167impl RollingBaselineDetector {
168    /// Default capacity (32) and threshold (0.5).
169    #[must_use]
170    pub fn new() -> Self {
171        Self {
172            history: std::sync::Mutex::new(std::collections::HashMap::new()),
173            capacity: 32,
174            byte_collapse_threshold: 0.5,
175        }
176    }
177
178    /// Override the per-identity ring-buffer cap.
179    #[must_use]
180    pub const fn with_capacity(mut self, capacity: usize) -> Self {
181        self.capacity = capacity;
182        self
183    }
184
185    /// Override the `ByteCollapse` ratio threshold.
186    #[must_use]
187    pub const fn with_byte_collapse_threshold(mut self, threshold: f64) -> Self {
188        self.byte_collapse_threshold = threshold;
189        self
190    }
191
192    fn append_to_history(&self, observation: &ContentTypeObservation) {
193        let Ok(mut map) = self.history.lock() else {
194            return;
195        };
196        let entry = map
197            .entry(observation.key.clone())
198            .or_insert_with(|| VecDeque::with_capacity(self.capacity));
199        if entry.len() == self.capacity {
200            entry.pop_front();
201        }
202        entry.push_back(observation.clone());
203    }
204
205    fn baseline(&self, key: &str) -> Option<ContentTypeObservation> {
206        let map = self.history.lock().ok()?;
207        map.get(key).and_then(|q| q.back().cloned())
208    }
209}
210
211impl Default for RollingBaselineDetector {
212    fn default() -> Self {
213        Self::new()
214    }
215}
216
217impl ContentTypeShiftDetector for RollingBaselineDetector {
218    fn record(
219        &mut self,
220        observation: &ContentTypeObservation,
221    ) -> Result<ContentTypeShiftReport, ContentTypeError> {
222        if observation.key.is_empty() {
223            return Err(ContentTypeError::EmptyKey);
224        }
225
226        let current_class = MimeClass::from_content_type(&observation.content_type);
227        let baseline = self.baseline(&observation.key);
228        let drift = baseline.as_ref().and_then(|baseline| {
229            let baseline_class = MimeClass::from_content_type(&baseline.content_type);
230            if baseline_class != current_class {
231                Some(ContentTypeDrift::ClassChange {
232                    from: baseline_class,
233                    to: current_class,
234                })
235            } else if baseline.byte_length > 0 {
236                // `u64 as f64` may lose precision for bodies > 2^53
237                // bytes (~9 PB). HTTP response bodies are capped well
238                // below that, so the precision loss is acceptable.
239                #[allow(clippy::cast_precision_loss)]
240                let ratio = observation.byte_length as f64 / baseline.byte_length as f64;
241                if ratio < self.byte_collapse_threshold {
242                    Some(ContentTypeDrift::ByteCollapse { ratio })
243                } else {
244                    None
245                }
246            } else {
247                None
248            }
249        });
250
251        self.append_to_history(observation);
252
253        Ok(ContentTypeShiftReport {
254            key: observation.key.clone(),
255            current: observation.clone(),
256            drift,
257        })
258    }
259}
260
261/// Helper for the snapshot builder: format a drift report as a short
262/// human-readable string suitable for inclusion in a diagnostic
263/// summary line.
264#[must_use]
265pub fn drift_summary(report: &ContentTypeShiftReport) -> String {
266    match &report.drift {
267        None => format!(
268            "[{}] {} (no drift)",
269            report.key, report.current.content_type
270        ),
271        Some(ContentTypeDrift::ClassChange { from, to }) => {
272            format!(
273                "[{}] class drift: {} -> {}",
274                report.key,
275                class_label(*from),
276                class_label(*to)
277            )
278        }
279        Some(ContentTypeDrift::ByteCollapse { ratio }) => {
280            format!("[{}] byte collapse: ratio={:.3}", report.key, ratio)
281        }
282    }
283}
284
285const fn class_label(class: MimeClass) -> &'static str {
286    match class {
287        MimeClass::Html => "html",
288        MimeClass::Json => "json",
289        MimeClass::Xml => "xml",
290        MimeClass::Markdown => "markdown",
291        MimeClass::Text => "text",
292        MimeClass::Binary => "binary",
293        MimeClass::Unknown => "unknown",
294    }
295}
296
297/// Unix-seconds timestamp helper for callers that need to record the
298/// observation's wall-clock time alongside the drift event.
299#[must_use]
300pub fn current_unix_secs() -> u64 {
301    SystemTime::now()
302        .duration_since(UNIX_EPOCH)
303        .map_or(0, |d| d.as_secs())
304}
305
306#[cfg(test)]
307#[allow(
308    clippy::unwrap_used,
309    clippy::expect_used,
310    clippy::panic,
311    clippy::indexing_slicing
312)]
313mod tests {
314    use super::*;
315
316    fn obs(key: &str, ct: &str, len: u64) -> ContentTypeObservation {
317        ContentTypeObservation {
318            key: key.to_string(),
319            content_type: ct.to_string(),
320            byte_length: len,
321        }
322    }
323
324    #[test]
325    fn mime_class_html_with_charset_param() {
326        assert_eq!(
327            MimeClass::from_content_type("text/html; charset=utf-8"),
328            MimeClass::Html
329        );
330    }
331
332    #[test]
333    fn mime_class_json_with_vendor_suffix() {
334        assert_eq!(
335            MimeClass::from_content_type("application/vnd.api+json"),
336            MimeClass::Json
337        );
338    }
339
340    #[test]
341    fn mime_class_markdown_stub() {
342        assert_eq!(
343            MimeClass::from_content_type("text/markdown; charset=utf-8"),
344            MimeClass::Markdown
345        );
346    }
347
348    #[test]
349    fn mime_class_xml() {
350        assert_eq!(
351            MimeClass::from_content_type("application/rss+xml"),
352            MimeClass::Xml
353        );
354    }
355
356    #[test]
357    fn mime_class_binary_octet_stream() {
358        assert_eq!(
359            MimeClass::from_content_type("application/octet-stream"),
360            MimeClass::Binary
361        );
362    }
363
364    #[test]
365    fn mime_class_unknown_garbage() {
366        assert_eq!(
367            MimeClass::from_content_type("not a mime type"),
368            MimeClass::Unknown
369        );
370        assert_eq!(MimeClass::from_content_type(""), MimeClass::Unknown);
371    }
372
373    #[test]
374    fn first_observation_records_no_drift() {
375        let mut d = RollingBaselineDetector::new();
376        let report = d
377            .record(&obs("https://example.com/a", "text/html", 100))
378            .unwrap();
379        assert_eq!(report.key, "https://example.com/a");
380        assert!(report.drift.is_none(), "first observation must not drift");
381    }
382
383    #[test]
384    fn two_identical_observations_no_drift() {
385        let mut d = RollingBaselineDetector::new();
386        d.record(&obs("k1", "text/html", 100)).unwrap();
387        let report = d.record(&obs("k1", "text/html", 110)).unwrap();
388        assert!(
389            report.drift.is_none(),
390            "identical observations must not drift"
391        );
392    }
393
394    #[test]
395    fn html_to_markdown_shift_is_class_change() {
396        let mut d = RollingBaselineDetector::new();
397        d.record(&obs("k1", "text/html", 50_000)).unwrap();
398        let report = d.record(&obs("k1", "text/markdown", 2_000)).unwrap();
399        match report.drift {
400            Some(ContentTypeDrift::ClassChange { from, to }) => {
401                assert_eq!(from, MimeClass::Html);
402                assert_eq!(to, MimeClass::Markdown);
403            }
404            other => panic!("expected ClassChange, got {other:?}"),
405        }
406    }
407
408    #[test]
409    fn massive_byte_collapse_emits_byte_collapse_drift() {
410        let mut d = RollingBaselineDetector::new();
411        d.record(&obs("k1", "text/html", 100_000)).unwrap();
412        let report = d.record(&obs("k1", "text/html", 4_000)).unwrap();
413        match report.drift {
414            Some(ContentTypeDrift::ByteCollapse { ratio }) => {
415                assert!(
416                    ratio > 0.0 && ratio < 0.5,
417                    "ratio must be below the 0.5 threshold, got {ratio}"
418                );
419            }
420            other => panic!("expected ByteCollapse, got {other:?}"),
421        }
422    }
423
424    #[test]
425    fn byte_collapse_threshold_is_configurable() {
426        // Threshold 0.9 → any drop below 90% triggers a collapse.
427        let mut d = RollingBaselineDetector::new().with_byte_collapse_threshold(0.9);
428        d.record(&obs("k1", "text/html", 1000)).unwrap();
429        let report = d.record(&obs("k1", "text/html", 200)).unwrap();
430        assert!(
431            matches!(report.drift, Some(ContentTypeDrift::ByteCollapse { .. })),
432            "80% drop must collapse when threshold is 0.9"
433        );
434
435        // Threshold 0.1 → any drop below 10% triggers a collapse.
436        // 95% drop (1000→50) has ratio 0.05, well below 0.1.
437        let mut d2 = RollingBaselineDetector::new().with_byte_collapse_threshold(0.1);
438        d2.record(&obs("k2", "text/html", 1000)).unwrap();
439        let report2 = d2.record(&obs("k2", "text/html", 50)).unwrap();
440        assert!(
441            matches!(report2.drift, Some(ContentTypeDrift::ByteCollapse { .. })),
442            "95% drop must collapse when threshold is 0.1"
443        );
444    }
445
446    #[test]
447    fn distinct_keys_dont_cross_pollute() {
448        let mut d = RollingBaselineDetector::new();
449        d.record(&obs("k1", "text/html", 50_000)).unwrap();
450        let report = d.record(&obs("k2", "text/markdown", 100)).unwrap();
451        assert!(
452            report.drift.is_none(),
453            "different keys must not trigger drift"
454        );
455    }
456
457    #[test]
458    fn ring_buffer_caps_at_capacity() {
459        let mut d = RollingBaselineDetector::new().with_capacity(3);
460        for i in 0..10 {
461            d.record(&obs("k1", "text/html", 100 + i)).unwrap();
462        }
463        // Only the last 3 should be retained; the baseline should
464        // compare against the last-inserted (most recent), not the
465        // 0th. Verify by recording one more and checking the report
466        // is anchored to length 109 (the most recent prior).
467        let report = d.record(&obs("k1", "text/html", 150)).unwrap();
468        assert!(report.drift.is_none(), "100→150 is growth, not collapse");
469    }
470
471    #[test]
472    fn record_rejects_empty_key() {
473        let mut d = RollingBaselineDetector::new();
474        let err = d
475            .record(&obs("", "text/html", 100))
476            .expect_err("empty key must be rejected");
477        assert!(matches!(err, ContentTypeError::EmptyKey));
478    }
479
480    #[test]
481    fn drift_summary_strings_have_useful_content() {
482        let mut d = RollingBaselineDetector::new();
483        d.record(&obs("k1", "text/html", 50_000)).unwrap();
484        let report = d.record(&obs("k1", "text/markdown", 100)).unwrap();
485        let s = drift_summary(&report);
486        assert!(s.contains("k1"));
487        assert!(s.contains("html"));
488        assert!(s.contains("markdown"));
489    }
490
491    #[test]
492    fn zero_baseline_does_not_divide_by_zero() {
493        let mut d = RollingBaselineDetector::new();
494        d.record(&obs("k1", "text/html", 0)).unwrap();
495        let report = d.record(&obs("k1", "text/html", 100)).unwrap();
496        assert!(
497            report.drift.is_none(),
498            "zero-byte baseline must not emit ByteCollapse (no division)"
499        );
500    }
501}