Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_text/tests/detect_corpus/python/canonical.txt

1.8 KiB, 1 run

created by r1870400018:12022, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1from collections import defaultdict
2from typing import Dict, List, Optional
3import json
4import sys
5
6
7class WordIndex:
8 """Index of word positions within a document."""
9
10 def __init__(self):
11 self.index: Dict[str, List[int]] = defaultdict(list)
12 self.doc_count = 0
13
14 def add_document(self, doc_id: int, text: str) -> None:
15 words = text.lower().split()
16 for pos, word in enumerate(words):
17 cleaned = ''.join(c for c in word if c.isalnum())
18 if cleaned:
19 self.index[cleaned].append((doc_id, pos))
20 self.doc_count += 1
21
22 def search(self, term: str) -> List[tuple]:
23 return self.index.get(term.lower(), [])
24
25 def term_count(self) -> int:
26 return len(self.index)
27
28 def document_frequency(self, term: str) -> int:
29 hits = self.search(term)
30 return len(set(doc_id for doc_id, _ in hits))
31
32 def to_json(self) -> str:
33 return json.dumps({
34 'doc_count': self.doc_count,
35 'term_count': self.term_count(),
36 'index': {k: v for k, v in self.index.items()},
37 }, indent=2)
38
39
40def build_index(documents: Dict[int, str]) -> WordIndex:
41 idx = WordIndex()
42 for doc_id, text in documents.items():
43 idx.add_document(doc_id, text)
44 return idx
45
46
47if __name__ == '__main__':
48 docs = {
49 1: "the quick brown fox jumps over the lazy dog",
50 2: "the fox went to the market to buy some eggs",
51 3: "a lazy dog slept in the sun all afternoon",
52 }
53
54 index = build_index(docs)
55 print(f"indexed {index.doc_count} documents, {index.term_count()} terms")
56
57 for term in sys.argv[1:]:
58 hits = index.search(term)
59 df = index.document_frequency(term)
60 print(f" '{term}': {len(hits)} occurrences in {df} documents")