Spaces:
Paused
Paused
File size: 5,831 Bytes
5e0b58b | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 | """Validation-only knowledge intake boundary.
Network access is deliberately absent in alpha.6. Future Wikipedia or web adapters must
produce SourceRecord + KnowledgeItem values before any downstream learning stimulus is
created.
This module provides the KnowledgeIntakeValidator, which creates provenance-bearing
KnowledgeItem instances from raw content drafts. All validation is performed before
any item enters the knowledge pipeline.
"""
from __future__ import annotations
import hashlib
import time
from dataclasses import dataclass
from typing import Any
from .models import KnowledgeItem, SourceRecord
# ============================================================================
# Knowledge Draft
# ============================================================================
@dataclass(frozen=True, slots=True)
class KnowledgeDraft:
"""Raw knowledge content before validation and ingestion.
Attributes:
source_type: Type of source (e.g., 'wikipedia', 'web', 'document').
locator: Source identifier (URL, file path, DOI, etc.).
title: Title or heading of the content.
content: The raw content text.
language: ISO 639-1 language code (default: 'und' for undetermined).
confidence: Confidence score between 0.0 and 1.0.
"""
source_type: str
locator: str
title: str
content: str
language: str = "und"
confidence: float = 0.0
mime_type: str = "text/plain"
source_version: str = "not_reported"
trust_classification: str = "UNKNOWN"
extraction_method: str = "not_reported"
def to_dict(self) -> dict[str, Any]:
"""Convert to dictionary for serialization."""
return {
"source_type": self.source_type,
"locator": self.locator,
"title": self.title,
"content": self.content,
"language": self.language,
"confidence": self.confidence,
"mime_type": self.mime_type,
"source_version": self.source_version,
"trust_classification": self.trust_classification,
"extraction_method": self.extraction_method,
}
# ============================================================================
# Knowledge Intake Validator
# ============================================================================
class KnowledgeIntakeValidator:
"""Create provenance-bearing items from already retrieved content.
This validator ensures that all incoming knowledge meets quality standards
before being converted into KnowledgeItem instances. It creates a
cryptographic content hash for provenance tracking.
Example:
>>> validator = KnowledgeIntakeValidator()
>>> draft = KnowledgeDraft(
... source_type="wikipedia",
... locator="https://en.wikipedia.org/wiki/Paris",
... title="Paris",
... content="Paris is the capital of France."
... )
>>> item = validator.create_item(
... item_id="item_001",
... source_id="src_001",
... draft=draft
... )
>>> print(item.content)
Paris is the capital of France.
"""
def create_item(
self,
*,
item_id: str,
source_id: str,
draft: KnowledgeDraft,
) -> KnowledgeItem:
"""Create a provenance-bearing KnowledgeItem from a draft.
Args:
item_id: Unique identifier for the knowledge item.
source_id: Unique identifier for the source.
draft: The raw knowledge draft to validate and convert.
Returns:
A fully validated KnowledgeItem with provenance.
Raises:
ValueError: If the content is empty, confidence is out of bounds,
or required fields are missing.
"""
captured_at_ns = time.time_ns()
# Validate content
if not draft.content or not draft.content.strip():
raise ValueError("knowledge content must not be empty")
# Validate confidence
if not 0.0 <= draft.confidence <= 1.0:
raise ValueError(
f"confidence must be between 0 and 1, got {draft.confidence}"
)
# Validate title
if not draft.title or not draft.title.strip():
raise ValueError("knowledge title must not be empty")
# Validate source_type
if not draft.source_type or not draft.source_type.strip():
raise ValueError("source_type must not be empty")
# Create content hash
digest = hashlib.sha256(draft.content.encode("utf-8")).hexdigest()
# Create SourceRecord with provenance
processed_at_ns = time.time_ns()
source = SourceRecord(
source_id=source_id,
source_type=draft.source_type,
locator=draft.locator,
retrieved_at_ns=time.time_ns(),
content_sha256=digest,
captured_at_ns=captured_at_ns,
processed_at_ns=processed_at_ns,
mime_type=draft.mime_type,
source_version=draft.source_version,
trust_classification=draft.trust_classification,
extraction_method=draft.extraction_method,
)
# Create KnowledgeItem
return KnowledgeItem(
item_id=item_id,
source=source,
title=draft.title,
content=draft.content,
language=draft.language,
confidence=draft.confidence,
)
# ============================================================================
# Module Exports
# ============================================================================
__all__ = [
"KnowledgeDraft",
"KnowledgeIntakeValidator",
"KnowledgeItem",
"SourceRecord",
]
|