test build
Some checks failed
Deploy / deploy (push) Has been cancelled

This commit is contained in:
jze9
2026-05-18 01:14:40 +05:00
commit 2a14350ee3
46 changed files with 3620 additions and 0 deletions

View File

@@ -0,0 +1,149 @@
import re
from typing import Dict, List, Any, Tuple
from app.core.fingerprint import WinNowing, MinHashLSHIndex
from app.core.embeddings import EmbeddingIndex
from app.core.fulltext import FulltextSearch
from app.config import settings
FRAGMENT_WORDS = 40
def _split_sentences(text: str) -> List[str]:
parts = re.split(r'(?<=[.!?])\s+', text.strip())
return [p for p in parts if p]
def _extract_fragment(text: str, sentence_idx: int, context_sentences: int = 2) -> Tuple[str, int, int]:
"""Return (fragment_text, char_start, char_end) around the given sentence index."""
sentences = _split_sentences(text)
start_i = max(0, sentence_idx - context_sentences)
end_i = min(len(sentences), sentence_idx + context_sentences + 1)
fragment = " ".join(sentences[start_i:end_i])
char_start = text.find(sentences[start_i]) if sentences else 0
char_end = char_start + len(fragment)
return fragment, char_start, char_end
def _truncate(text: str, max_words: int = FRAGMENT_WORDS) -> str:
words = text.split()
if len(words) <= max_words:
return text
return " ".join(words[:max_words]) + ""
class DocumentAnalyzer:
def __init__(self):
self.winnowing = WinNowing()
self.minhash_lsh = MinHashLSHIndex(threshold=settings.FINGERPRINT_THRESHOLD)
self.embeddings = EmbeddingIndex()
self.fulltext = FulltextSearch()
def analyze(self, text: str, task_id: str) -> Dict[str, Any]:
matches: List[Dict[str, Any]] = []
matches.extend(self._level_1_exact_match(text))
matches.extend(self._level_2_fuzzy_match(text))
matches.extend(self._level_3_semantic_match(text))
return {
"task_id": task_id,
"status": "completed",
"overall_similarity": self._calculate_overall_similarity(matches),
"matches": matches[:settings.TOP_K_RESULTS],
}
def _level_1_exact_match(self, text: str, top_k: int = 5) -> List[Dict[str, Any]]:
sentences = _split_sentences(text)
if not sentences:
return []
results = []
for idx, sentence in enumerate(sentences[:20]):
if len(sentence.split()) < 6:
continue
try:
es_results = self.fulltext.search_phrase(sentence, top_k=2)
for r in es_results:
r["_sentence_idx"] = idx
results.extend(es_results)
except Exception:
continue
matches = []
seen_docs: set = set()
for result in results:
if result["doc_id"] in seen_docs:
continue
idx = result.get("_sentence_idx", 0)
fragment, char_start, char_end = _extract_fragment(text, idx)
matches.append({
"method": "exact",
"similarity": min(result["score"] * 20, 100),
"source_id": str(result["doc_id"]), # передаём doc_id напрямую
"source_title": result["title"],
"source_db": result["source"],
"url": result["url"],
"fragment_text": _truncate(fragment),
"fragment_start": char_start,
"fragment_end": char_end,
})
seen_docs.add(result["doc_id"])
return matches[:top_k]
def _level_2_fuzzy_match(self, text: str, top_k: int = 5) -> List[Dict[str, Any]]:
try:
fuzzy_results = self.minhash_lsh.query(text, top_k=top_k)
except Exception:
return []
sentences = _split_sentences(text)
fragment = _truncate(sentences[0]) if sentences else ""
matches = []
for doc_id, similarity in fuzzy_results:
if similarity >= settings.FINGERPRINT_THRESHOLD:
matches.append({
"method": "fuzzy",
"similarity": similarity * 100,
"source_id": doc_id,
"fragment_text": fragment,
"fragment_start": 0,
"fragment_end": len(fragment),
})
return matches
def _level_3_semantic_match(self, text: str, top_k: int = 5) -> List[Dict[str, Any]]:
try:
embedding_results = self.embeddings.search(text, top_k=top_k)
except Exception:
return []
sentences = _split_sentences(text)
fragment = _truncate(sentences[0]) if sentences else ""
matches = []
for doc_id, similarity in embedding_results:
if similarity >= settings.SEMANTIC_THRESHOLD:
matches.append({
"method": "semantic",
"similarity": similarity * 100,
"source_id": doc_id,
"fragment_text": fragment,
"fragment_start": 0,
"fragment_end": len(fragment),
})
return matches
def _calculate_overall_similarity(self, matches: List[Dict[str, Any]]) -> float:
if not matches:
return 0.0
weights = {"exact": 0.4, "fuzzy": 0.3, "semantic": 0.3}
total_weight = 0.0
total_similarity = 0.0
for match in matches:
w = weights.get(match["method"], 0.1)
total_similarity += match["similarity"] * w
total_weight += w
return round(total_similarity / total_weight if total_weight > 0 else 0, 1)