"""Tests for retrieval-based novelty and duplicate detection. Validates: - Exact fingerprint consistency - SimHash near-duplicate detection - Embedding backend returns correct dimensions - Cosine similarity bounds - Index search returns sorted results - Novelty formula returns [0, 1] range - Duplicate document gets low novelty - Novel document gets high novelty """ from __future__ import annotations import pytest from services.intelligence_pipeline_v3.novelty.embeddings import ( MockEmbeddingBackend, SentenceTransformerBackend, cosine_similarity, ) from services.intelligence_pipeline_v3.novelty.fingerprints import ( compute_exact_fingerprint, compute_simhash, hamming_distance, is_near_duplicate, ) from services.intelligence_pipeline_v3.novelty.index import NoveltyIndex from services.intelligence_pipeline_v3.novelty.scorer import NoveltyScorer # --- Fingerprint tests --- class TestExactFingerprint: """Test exact fingerprint consistency.""" def test_same_text_same_fingerprint(self) -> None: """Identical text always produces the same fingerprint.""" text = "Apple reports record quarterly revenue of $94.8 billion" fp1 = compute_exact_fingerprint(text) fp2 = compute_exact_fingerprint(text) assert fp1 == fp2 def test_normalized_whitespace(self) -> None: """Different whitespace patterns produce the same fingerprint.""" text1 = "Apple reports record quarterly revenue" text2 = "Apple reports record quarterly revenue" text3 = "Apple\treports\nrecord\tquarterly revenue" assert compute_exact_fingerprint(text1) == compute_exact_fingerprint(text2) assert compute_exact_fingerprint(text1) == compute_exact_fingerprint(text3) def test_case_insensitive(self) -> None: """Case differences produce the same fingerprint.""" text1 = "Apple Reports Record Quarterly Revenue" text2 = "apple reports record quarterly revenue" assert compute_exact_fingerprint(text1) == compute_exact_fingerprint(text2) def test_different_text_different_fingerprint(self) -> None: """Meaningfully different text produces different fingerprints.""" fp1 = compute_exact_fingerprint("Apple reports record revenue") fp2 = compute_exact_fingerprint("Google reports declining revenue") assert fp1 != fp2 def test_fingerprint_is_hex_sha256(self) -> None: """Fingerprint is a valid 64-char hex SHA-256 digest.""" fp = compute_exact_fingerprint("test content") assert len(fp) == 64 assert all(c in "0123456789abcdef" for c in fp) def test_empty_text(self) -> None: """Empty text produces a valid fingerprint.""" fp = compute_exact_fingerprint("") assert len(fp) == 64 # Empty and whitespace-only should match after normalization assert fp == compute_exact_fingerprint(" ") class TestSimhashNearDuplicate: """Test SimHash near-duplicate detection.""" def test_identical_text_zero_distance(self) -> None: """Identical text has hamming distance 0.""" text = "Apple reports record quarterly revenue of $94.8 billion" sh1 = compute_simhash(text) sh2 = compute_simhash(text) assert hamming_distance(sh1, sh2) == 0 def test_similar_text_lower_distance_than_unrelated(self) -> None: """Text with minor edits has lower distance than completely unrelated text.""" text1 = "Apple reports record quarterly revenue of $94.8 billion dollars" text2 = "Apple reports record quarterly revenue of $94.8 billion usd" text3 = "The weather in Tokyo is sunny with temperatures around 25 degrees" sh1 = compute_simhash(text1) sh2 = compute_simhash(text2) sh3 = compute_simhash(text3) # Similar texts should have lower distance than unrelated texts assert hamming_distance(sh1, sh2) < hamming_distance(sh1, sh3) def test_different_text_high_distance(self) -> None: """Completely different text should have higher distance.""" text1 = "Apple reports record quarterly revenue of $94.8 billion" text2 = "The weather in Tokyo is sunny with temperatures around 25 degrees" sh1 = compute_simhash(text1) sh2 = compute_simhash(text2) # Very different content should produce measurable distance assert hamming_distance(sh1, sh2) > 5 def test_is_near_duplicate_true(self) -> None: """Near-duplicate detection returns True for identical content.""" text = "The Federal Reserve raised interest rates by 25 basis points today" sh1 = compute_simhash(text) sh2 = compute_simhash(text) # Identical text has distance 0, always a near-duplicate assert is_near_duplicate(sh1, sh2) is True assert hamming_distance(sh1, sh2) == 0 def test_is_near_duplicate_false_for_unrelated(self) -> None: """Near-duplicate detection returns False for unrelated documents.""" text1 = "Apple reports record quarterly revenue of $94.8 billion" text2 = "The weather in Tokyo is sunny with temperatures around 25 degrees" sh1 = compute_simhash(text1) sh2 = compute_simhash(text2) # Very different content should not be near-duplicate at threshold 3 # (depends on content but unlikely to collide) distance = hamming_distance(sh1, sh2) assert distance > 3 or not is_near_duplicate(sh1, sh2, threshold=2) def test_hamming_distance_bounds(self) -> None: """Hamming distance is always between 0 and 64 for 64-bit hashes.""" sh1 = compute_simhash("test text one") sh2 = compute_simhash("different content entirely here") dist = hamming_distance(sh1, sh2) assert 0 <= dist <= 64 def test_hamming_distance_symmetric(self) -> None: """Hamming distance is symmetric: d(a,b) == d(b,a).""" sh1 = compute_simhash("first document") sh2 = compute_simhash("second document") assert hamming_distance(sh1, sh2) == hamming_distance(sh2, sh1) def test_empty_text_simhash(self) -> None: """Empty text produces a simhash of 0.""" assert compute_simhash("") == 0 assert compute_simhash(" ") == 0 def test_custom_threshold(self) -> None: """Custom threshold adjusts near-duplicate sensitivity.""" sh1 = 0b1111111111111111111111111111111111111111111111111111111111111111 sh2 = 0b1111111111111111111111111111111111111111111111111111111111111110 # Distance is 1 assert is_near_duplicate(sh1, sh2, threshold=1) is True assert is_near_duplicate(sh1, sh2, threshold=0) is False # --- Embedding backend tests --- class TestEmbeddingBackend: """Test embedding backend returns correct dimensions.""" def test_mock_backend_correct_dimension(self) -> None: """MockEmbeddingBackend produces vectors of specified dimension.""" backend = MockEmbeddingBackend(dimension=384) texts = ["Test sentence one", "Test sentence two"] embeddings = backend.embed(texts) assert len(embeddings) == 2 assert all(len(e) == 384 for e in embeddings) def test_mock_backend_custom_dimension(self) -> None: """MockEmbeddingBackend respects custom dimension.""" backend = MockEmbeddingBackend(dimension=128) embeddings = backend.embed(["hello world"]) assert len(embeddings[0]) == 128 def test_mock_backend_deterministic(self) -> None: """Same text always produces the same embedding.""" backend = MockEmbeddingBackend(dimension=384) text = "Apple reports revenue" e1 = backend.embed([text]) e2 = backend.embed([text]) assert e1 == e2 def test_mock_backend_different_texts_different_embeddings(self) -> None: """Different texts produce different embeddings.""" backend = MockEmbeddingBackend(dimension=384) embeddings = backend.embed(["Apple revenue", "Google revenue"]) assert embeddings[0] != embeddings[1] def test_mock_backend_unit_normalized(self) -> None: """MockEmbeddingBackend produces approximately unit-normalized vectors.""" import math backend = MockEmbeddingBackend(dimension=384) embeddings = backend.embed(["test text"]) norm = math.sqrt(sum(x * x for x in embeddings[0])) assert abs(norm - 1.0) < 1e-6 def test_sentence_transformer_dimension_property(self) -> None: """SentenceTransformerBackend declares 384 dimensions.""" backend = SentenceTransformerBackend() assert backend.dimension == 384 def test_empty_text_embedding(self) -> None: """Empty string can be embedded without error.""" backend = MockEmbeddingBackend(dimension=384) embeddings = backend.embed([""]) assert len(embeddings) == 1 assert len(embeddings[0]) == 384 # --- Cosine similarity tests --- class TestCosineSimilarity: """Test cosine similarity bounds.""" def test_identical_vectors(self) -> None: """Identical vectors have similarity 1.0.""" v = [1.0, 2.0, 3.0] assert abs(cosine_similarity(v, v) - 1.0) < 1e-9 def test_opposite_vectors(self) -> None: """Opposite vectors have similarity -1.0.""" v1 = [1.0, 0.0, 0.0] v2 = [-1.0, 0.0, 0.0] assert abs(cosine_similarity(v1, v2) - (-1.0)) < 1e-9 def test_orthogonal_vectors(self) -> None: """Orthogonal vectors have similarity 0.0.""" v1 = [1.0, 0.0, 0.0] v2 = [0.0, 1.0, 0.0] assert abs(cosine_similarity(v1, v2)) < 1e-9 def test_similarity_in_bounds(self) -> None: """Cosine similarity is always in [-1, 1].""" backend = MockEmbeddingBackend(dimension=64) texts = ["apple", "banana", "cherry", "date"] embeddings = backend.embed(texts) for i in range(len(embeddings)): for j in range(len(embeddings)): sim = cosine_similarity(embeddings[i], embeddings[j]) assert -1.0 - 1e-9 <= sim <= 1.0 + 1e-9 def test_zero_vector(self) -> None: """Zero vector returns similarity 0.0.""" v1 = [0.0, 0.0, 0.0] v2 = [1.0, 2.0, 3.0] assert cosine_similarity(v1, v2) == 0.0 def test_dimension_mismatch_raises(self) -> None: """Mismatched dimensions raise ValueError.""" v1 = [1.0, 2.0] v2 = [1.0, 2.0, 3.0] with pytest.raises(ValueError, match="same dimension"): cosine_similarity(v1, v2) # --- Index search tests --- class TestNoveltyIndex: """Test index search returns sorted results.""" def test_search_returns_sorted_by_similarity(self) -> None: """Search results are sorted descending by similarity.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() texts = ["apple stock", "banana fruit", "cherry pie", "apple revenue"] embeddings = backend.embed(texts) for i, (text, emb) in enumerate(zip(texts, embeddings)): index.add(f"doc_{i}", emb, {"text": text}) # Query with something similar to "apple stock" query = embeddings[0] results = index.search(query, k=4) # Results should be sorted descending for i in range(len(results) - 1): assert results[i].similarity_score >= results[i + 1].similarity_score def test_search_top_match_is_self(self) -> None: """Searching with an indexed embedding returns itself as top match.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() emb = backend.embed(["test document"])[0] index.add("doc_1", emb) results = index.search(emb, k=1) assert len(results) == 1 assert results[0].doc_id == "doc_1" assert results[0].similarity_score > 0.99 def test_search_respects_k(self) -> None: """Search returns at most k results.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() for i in range(10): emb = backend.embed([f"document {i}"])[0] index.add(f"doc_{i}", emb) query = backend.embed(["document 0"])[0] results = index.search(query, k=3) assert len(results) == 3 def test_search_empty_index(self) -> None: """Searching an empty index returns empty results.""" index = NoveltyIndex() results = index.search([0.1] * 64, k=5) assert results == [] def test_add_and_update(self) -> None: """Adding with an existing doc_id updates the embedding.""" index = NoveltyIndex() index.add("doc_1", [1.0, 0.0, 0.0]) index.add("doc_1", [0.0, 1.0, 0.0]) assert len(index) == 1 results = index.search([0.0, 1.0, 0.0], k=1) assert results[0].doc_id == "doc_1" assert results[0].similarity_score > 0.99 def test_remove(self) -> None: """Removing a document excludes it from search.""" index = NoveltyIndex() index.add("doc_1", [1.0, 0.0, 0.0]) index.add("doc_2", [0.0, 1.0, 0.0]) assert len(index) == 2 index.remove("doc_1") assert len(index) == 1 results = index.search([1.0, 0.0, 0.0], k=5) assert all(r.doc_id != "doc_1" for r in results) def test_similarity_scores_clamped(self) -> None: """Similarity scores are clamped to [0, 1].""" index = NoveltyIndex() index.add("doc_1", [1.0, 0.0, 0.0]) index.add("doc_2", [-1.0, 0.0, 0.0]) results = index.search([1.0, 0.0, 0.0], k=2) for r in results: assert 0.0 <= r.similarity_score <= 1.0 # --- Novelty formula tests --- class TestNoveltyScorer: """Test novelty formula returns [0, 1] range.""" def test_novelty_in_range(self) -> None: """All novelty scores are in [0, 1].""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() # Add some documents to the index for i in range(5): emb = backend.embed([f"existing document number {i}"])[0] index.add(f"existing_{i}", emb) # Score a new document doc_emb = backend.embed(["new document about technology"])[0] event_emb = backend.embed(["tech earnings beat"])[0] scorer = NoveltyScorer(k=3) result = scorer.compute_novelty(doc_emb, event_emb, index) assert 0.0 <= result.document_novelty <= 1.0 assert 0.0 <= result.event_novelty <= 1.0 assert 0.0 <= result.combined_novelty <= 1.0 def test_duplicate_gets_low_novelty(self) -> None: """An exact duplicate document gets low novelty scores.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() # Index a document text = "Apple reports record quarterly revenue of $94.8 billion" emb = backend.embed([text])[0] index.add("original_doc", emb) # Score the same document scorer = NoveltyScorer(k=5) result = scorer.compute_novelty(emb, emb, index) # Should have very low novelty (embedding matches itself) assert result.document_novelty < 0.1 assert result.event_novelty < 0.1 assert result.combined_novelty < 0.1 def test_novel_document_gets_high_novelty(self) -> None: """A document unlike anything in the index gets high novelty.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() # Index documents about one topic for i in range(5): emb = backend.embed([f"weather forecast for city {i} rain expected"])[0] index.add(f"weather_{i}", emb) # Score a completely different topic doc_emb = backend.embed(["semiconductor shortage impacts automotive production"])[0] event_emb = backend.embed(["chip supply constraint"])[0] scorer = NoveltyScorer(k=5) result = scorer.compute_novelty(doc_emb, event_emb, index) # Should have high novelty assert result.document_novelty > 0.5 assert result.event_novelty > 0.5 assert result.combined_novelty > 0.5 def test_exact_duplicate_flag_forces_zero_novelty(self) -> None: """When is_exact_duplicate=True, novelty is 0.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() emb = backend.embed(["test"])[0] index.add("doc_1", emb) scorer = NoveltyScorer(k=5) result = scorer.compute_novelty(emb, emb, index, is_exact_duplicate=True) assert result.document_novelty == 0.0 assert result.event_novelty == 0.0 assert result.combined_novelty == 0.0 assert result.is_exact_duplicate is True def test_near_duplicate_flag_caps_novelty(self) -> None: """Near-duplicate flag caps document novelty at 0.2.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() # Index something mildly related index.add("doc_1", backend.embed(["somewhat related content"])[0]) doc_emb = backend.embed(["quite different content here"])[0] event_emb = backend.embed(["different event"])[0] scorer = NoveltyScorer(k=5) result = scorer.compute_novelty(doc_emb, event_emb, index, is_near_duplicate=True) assert result.document_novelty <= 0.2 assert result.is_near_duplicate is True def test_empty_index_full_novelty(self) -> None: """Empty index (no history) returns full novelty.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() doc_emb = backend.embed(["brand new content"])[0] event_emb = backend.embed(["new event"])[0] scorer = NoveltyScorer(k=5) result = scorer.compute_novelty(doc_emb, event_emb, index) assert result.document_novelty == 1.0 assert result.event_novelty == 1.0 assert result.combined_novelty == 1.0 def test_formula_version_tracked(self) -> None: """Result includes the formula version used.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() emb = backend.embed(["test"])[0] scorer = NoveltyScorer(k=5, formula_version="v1.0") result = scorer.compute_novelty(emb, emb, index) assert result.formula_version == "v1.0" def test_nearest_matches_included(self) -> None: """Result includes nearest matches for explainability.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() for i in range(3): emb = backend.embed([f"document {i}"])[0] index.add(f"doc_{i}", emb) query_emb = backend.embed(["document 0"])[0] scorer = NoveltyScorer(k=5) result = scorer.compute_novelty(query_emb, query_emb, index) assert len(result.nearest_matches) > 0 # Matches should be sorted by similarity descending for i in range(len(result.nearest_matches) - 1): assert ( result.nearest_matches[i].similarity_score >= result.nearest_matches[i + 1].similarity_score ) def test_combined_novelty_is_minimum(self) -> None: """Combined novelty is the minimum of document and event novelty.""" backend = MockEmbeddingBackend(dimension=64) index = NoveltyIndex() # Add a document similar to our test doc doc_emb = backend.embed(["known document"])[0] index.add("existing", doc_emb) # Query with something similar to doc but different event event_emb = backend.embed(["completely new event topic"])[0] scorer = NoveltyScorer(k=5) result = scorer.compute_novelty(doc_emb, event_emb, index) assert result.combined_novelty <= result.document_novelty assert result.combined_novelty <= result.event_novelty assert result.combined_novelty == min(result.document_novelty, result.event_novelty)