oxedyne/fe2o3/fe2o3_text/tests/detect_corpus/python/canonical.txt
1.8 KiB, 1 run
created by r1870400018:12022, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | from collections import defaultdict |
| 2 | from typing import Dict, List, Optional |
| 3 | import json |
| 4 | import sys |
| 5 | |
| 6 | |
| 7 | class WordIndex: |
| 8 | """Index of word positions within a document.""" |
| 9 | |
| 10 | def __init__(self): |
| 11 | self.index: Dict[str, List[int]] = defaultdict(list) |
| 12 | self.doc_count = 0 |
| 13 | |
| 14 | def add_document(self, doc_id: int, text: str) -> None: |
| 15 | words = text.lower().split() |
| 16 | for pos, word in enumerate(words): |
| 17 | cleaned = ''.join(c for c in word if c.isalnum()) |
| 18 | if cleaned: |
| 19 | self.index[cleaned].append((doc_id, pos)) |
| 20 | self.doc_count += 1 |
| 21 | |
| 22 | def search(self, term: str) -> List[tuple]: |
| 23 | return self.index.get(term.lower(), []) |
| 24 | |
| 25 | def term_count(self) -> int: |
| 26 | return len(self.index) |
| 27 | |
| 28 | def document_frequency(self, term: str) -> int: |
| 29 | hits = self.search(term) |
| 30 | return len(set(doc_id for doc_id, _ in hits)) |
| 31 | |
| 32 | def to_json(self) -> str: |
| 33 | return json.dumps({ |
| 34 | 'doc_count': self.doc_count, |
| 35 | 'term_count': self.term_count(), |
| 36 | 'index': {k: v for k, v in self.index.items()}, |
| 37 | }, indent=2) |
| 38 | |
| 39 | |
| 40 | def build_index(documents: Dict[int, str]) -> WordIndex: |
| 41 | idx = WordIndex() |
| 42 | for doc_id, text in documents.items(): |
| 43 | idx.add_document(doc_id, text) |
| 44 | return idx |
| 45 | |
| 46 | |
| 47 | if __name__ == '__main__': |
| 48 | docs = { |
| 49 | 1: "the quick brown fox jumps over the lazy dog", |
| 50 | 2: "the fox went to the market to buy some eggs", |
| 51 | 3: "a lazy dog slept in the sun all afternoon", |
| 52 | } |
| 53 | |
| 54 | index = build_index(docs) |
| 55 | print(f"indexed {index.doc_count} documents, {index.term_count()} terms") |
| 56 | |
| 57 | for term in sys.argv[1:]: |
| 58 | hits = index.search(term) |
| 59 | df = index.document_frequency(term) |
| 60 | print(f" '{term}': {len(hits)} occurrences in {df} documents") |