初始化项目,由ModelHub XC社区提供模型
Model: Italianhype/Blum-Finance-4B Source: Original Platform
This commit is contained in:
11
blum_finance/__init__.py
Normal file
11
blum_finance/__init__.py
Normal file
@@ -0,0 +1,11 @@
|
||||
from .inference import BlumFinancePipeline
|
||||
from .memory import BlumFinanceMemoryStore, InvalidMemoryRecord
|
||||
from .schemas import FinancialReasoningRequest, FinancialReasoningResponse
|
||||
|
||||
__all__ = [
|
||||
"BlumFinancePipeline",
|
||||
"BlumFinanceMemoryStore",
|
||||
"InvalidMemoryRecord",
|
||||
"FinancialReasoningRequest",
|
||||
"FinancialReasoningResponse",
|
||||
]
|
||||
237
blum_finance/contributions.py
Normal file
237
blum_finance/contributions.py
Normal file
@@ -0,0 +1,237 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from dataclasses import dataclass
|
||||
from datetime import UTC, datetime
|
||||
import hashlib
|
||||
import hmac
|
||||
import json
|
||||
from pathlib import Path
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
|
||||
TARGET_REPOSITORY = "Italianhype/Blum-Finance-Memory"
|
||||
BLOCKED_KEYS = {
|
||||
"access_token",
|
||||
"account_id",
|
||||
"api_key",
|
||||
"authorization",
|
||||
"broker_account_id",
|
||||
"refresh_token",
|
||||
}
|
||||
EMAIL_PATTERN = re.compile(r"\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b", re.IGNORECASE)
|
||||
HF_TOKEN_PATTERN = re.compile(r"\bhf_[A-Za-z0-9_]{8,}\b")
|
||||
|
||||
|
||||
class ConsentRequired(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ContributionBundleResult:
|
||||
path: Path
|
||||
content_hash: str
|
||||
uploaded: bool
|
||||
repository: str
|
||||
submission_url: str | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ContributionValidation:
|
||||
accepted: bool
|
||||
blockers: tuple[str, ...]
|
||||
status: str
|
||||
|
||||
|
||||
def build_contribution_bundle(
|
||||
payload: dict[str, Any],
|
||||
*,
|
||||
output: Path,
|
||||
consent: bool = False,
|
||||
push: bool = False,
|
||||
repository: str = TARGET_REPOSITORY,
|
||||
api: Any | None = None,
|
||||
) -> ContributionBundleResult:
|
||||
if not consent:
|
||||
raise ConsentRequired(
|
||||
"Community contribution is disabled until explicit consent is provided."
|
||||
)
|
||||
sanitized, redactions = _sanitize(payload)
|
||||
canonical = json.dumps(
|
||||
sanitized,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
)
|
||||
content_hash = hashlib.sha256(canonical.encode("utf-8")).hexdigest()
|
||||
bundle = {
|
||||
"schema_version": "blum-finance-contribution-v2",
|
||||
"content_hash": content_hash,
|
||||
"created_at": datetime.now(UTC).isoformat(),
|
||||
"target_repository": repository,
|
||||
"consent": {
|
||||
"explicit": True,
|
||||
"telemetry_default": "disabled",
|
||||
"license": "cc-by-4.0",
|
||||
},
|
||||
"redactions": redactions,
|
||||
"quarantine_status": "pending_validation",
|
||||
"payload": sanitized,
|
||||
}
|
||||
validation = validate_contribution_bundle(bundle)
|
||||
bundle["quarantine_status"] = validation.status
|
||||
bundle["validation_blockers"] = list(validation.blockers)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
output.write_text(
|
||||
json.dumps(bundle, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
uploaded = False
|
||||
submission_url = None
|
||||
if push:
|
||||
if api is None:
|
||||
from huggingface_hub import HfApi
|
||||
|
||||
api = HfApi()
|
||||
|
||||
result = api.upload_file(
|
||||
path_or_fileobj=str(output),
|
||||
path_in_repo=f"quarantine/{content_hash}.json",
|
||||
repo_id=repository,
|
||||
repo_type="dataset",
|
||||
commit_message=f"contrib: add quarantined example {content_hash[:12]}",
|
||||
create_pr=True,
|
||||
)
|
||||
uploaded = True
|
||||
submission_url = str(
|
||||
getattr(result, "pr_url", None)
|
||||
or getattr(result, "commit_url", None)
|
||||
or result
|
||||
)
|
||||
return ContributionBundleResult(
|
||||
path=output,
|
||||
content_hash=content_hash,
|
||||
uploaded=uploaded,
|
||||
repository=repository,
|
||||
submission_url=submission_url,
|
||||
)
|
||||
|
||||
|
||||
def validate_contribution_bundle(bundle: dict[str, Any]) -> ContributionValidation:
|
||||
blockers: list[str] = []
|
||||
payload = bundle.get("payload")
|
||||
if not isinstance(payload, dict):
|
||||
return ContributionValidation(False, ("payload_missing",), "rejected")
|
||||
expected_hash = hashlib.sha256(
|
||||
json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
||||
).hexdigest()
|
||||
if not hmac.compare_digest(str(bundle.get("content_hash") or ""), expected_hash):
|
||||
blockers.append("content_hash_mismatch")
|
||||
if (bundle.get("consent") or {}).get("explicit") is not True:
|
||||
blockers.append("explicit_consent_missing")
|
||||
request = payload.get("request")
|
||||
response = payload.get("response")
|
||||
outcome = payload.get("outcome")
|
||||
quality = payload.get("quality")
|
||||
if not isinstance(request, dict) or not request.get("evidence") or not request.get("as_of"):
|
||||
blockers.append("point_in_time_request_missing")
|
||||
if not isinstance(response, dict) or not response.get("thesis"):
|
||||
blockers.append("model_response_missing")
|
||||
if not isinstance(outcome, dict) or not outcome.get("observed_at"):
|
||||
blockers.append("mature_outcome_missing")
|
||||
else:
|
||||
try:
|
||||
decision_at = _parse_datetime((request or {}).get("as_of"))
|
||||
observed_at = _parse_datetime(outcome.get("observed_at"))
|
||||
if observed_at <= decision_at:
|
||||
blockers.append("outcome_chronology_invalid")
|
||||
except (TypeError, ValueError):
|
||||
blockers.append("outcome_timestamp_invalid")
|
||||
if str(outcome.get("status") or "").lower() in {"", "pending", "unresolved", "inconclusive"}:
|
||||
blockers.append("mature_outcome_missing")
|
||||
if not isinstance(quality, dict) or quality.get("source_verified") is not True:
|
||||
blockers.append("source_provenance_unverified")
|
||||
return ContributionValidation(
|
||||
accepted=not blockers,
|
||||
blockers=tuple(dict.fromkeys(blockers)),
|
||||
status="eligible_for_curation" if not blockers else "pending_validation",
|
||||
)
|
||||
|
||||
|
||||
def _sanitize(value: Any) -> tuple[Any, list[str]]:
|
||||
redactions: set[str] = set()
|
||||
|
||||
def clean(item: Any) -> Any:
|
||||
if isinstance(item, dict):
|
||||
result: dict[str, Any] = {}
|
||||
for raw_key, child in item.items():
|
||||
key = str(raw_key)
|
||||
if key.lower() in BLOCKED_KEYS:
|
||||
redactions.add(key.lower())
|
||||
continue
|
||||
result[key] = clean(child)
|
||||
return result
|
||||
if isinstance(item, list):
|
||||
return [clean(child) for child in item]
|
||||
if isinstance(item, str):
|
||||
text = EMAIL_PATTERN.sub(
|
||||
lambda _: _replace(redactions, "email", "[REDACTED_EMAIL]"),
|
||||
item,
|
||||
)
|
||||
return HF_TOKEN_PATTERN.sub(
|
||||
lambda _: _replace(redactions, "hugging_face_token", "[REDACTED_TOKEN]"),
|
||||
text,
|
||||
)
|
||||
return item
|
||||
|
||||
return clean(value), sorted(redactions)
|
||||
|
||||
|
||||
def _replace(redactions: set[str], label: str, replacement: str) -> str:
|
||||
redactions.add(label)
|
||||
return replacement
|
||||
|
||||
|
||||
def _parse_datetime(value: Any) -> datetime:
|
||||
parsed = datetime.fromisoformat(str(value).replace("Z", "+00:00"))
|
||||
if parsed.tzinfo is None:
|
||||
parsed = parsed.replace(tzinfo=UTC)
|
||||
return parsed.astimezone(UTC)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Create an explicit, redacted BLUM Finance contribution bundle."
|
||||
)
|
||||
parser.add_argument("input", type=Path)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
parser.add_argument("--consent", action="store_true")
|
||||
parser.add_argument("--push", action="store_true")
|
||||
parser.add_argument("--repository", default=TARGET_REPOSITORY)
|
||||
args = parser.parse_args()
|
||||
payload = json.loads(args.input.read_text(encoding="utf-8"))
|
||||
result = build_contribution_bundle(
|
||||
payload,
|
||||
output=args.output,
|
||||
consent=args.consent,
|
||||
push=args.push,
|
||||
repository=args.repository,
|
||||
)
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"path": str(result.path),
|
||||
"content_hash": result.content_hash,
|
||||
"uploaded": result.uploaded,
|
||||
"repository": result.repository,
|
||||
"submission_url": result.submission_url,
|
||||
},
|
||||
indent=2,
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
156
blum_finance/inference.py
Normal file
156
blum_finance/inference.py
Normal file
@@ -0,0 +1,156 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Callable, Literal
|
||||
|
||||
from pydantic import ValidationError
|
||||
|
||||
from .schemas import FinancialReasoningRequest, FinancialReasoningResponse
|
||||
from .memory import BlumFinanceMemoryStore
|
||||
|
||||
|
||||
SYSTEM_PROMPT = """You are BLUM Finance, an evidence-bound financial reasoning model.
|
||||
Use only the supplied point-in-time evidence. Separate supportive and contradictory
|
||||
evidence. Never invent prices, returns, events or sources. Return one JSON object that
|
||||
matches the requested schema. If evidence is insufficient, abstain explicitly."""
|
||||
|
||||
|
||||
class BlumFinancePipeline:
|
||||
def __init__(
|
||||
self,
|
||||
model_id: str = "Italianhype/Blum",
|
||||
*,
|
||||
revision: str | None = None,
|
||||
runtime: Literal["transformers", "mlx"] = "transformers",
|
||||
generator: Callable[[list[dict[str, str]]], str] | None = None,
|
||||
memory_store: BlumFinanceMemoryStore | None = None,
|
||||
memory_limit: int = 3,
|
||||
):
|
||||
self.model_id = model_id
|
||||
self.revision = revision
|
||||
self.runtime = runtime
|
||||
self._generator = generator
|
||||
self.memory_store = memory_store
|
||||
self.memory_limit = max(0, int(memory_limit))
|
||||
self._pipeline = None
|
||||
self._mlx_model = None
|
||||
self._mlx_tokenizer = None
|
||||
|
||||
def generate(
|
||||
self,
|
||||
request: FinancialReasoningRequest | dict,
|
||||
) -> FinancialReasoningResponse:
|
||||
parsed_request = (
|
||||
request
|
||||
if isinstance(request, FinancialReasoningRequest)
|
||||
else FinancialReasoningRequest.model_validate(request)
|
||||
)
|
||||
messages = [
|
||||
{"role": "system", "content": SYSTEM_PROMPT},
|
||||
]
|
||||
if self.memory_store is not None and self.memory_limit > 0:
|
||||
memories = self.memory_store.retrieve(parsed_request, limit=self.memory_limit)
|
||||
if memories:
|
||||
messages.append(
|
||||
{
|
||||
"role": "system",
|
||||
"content": (
|
||||
"Validated historical memory follows. It contains past analogies, "
|
||||
"not current market facts. Use it only to challenge the current thesis "
|
||||
"and never copy a past outcome into the present.\n"
|
||||
+ json.dumps(memories, ensure_ascii=False, sort_keys=True)
|
||||
),
|
||||
}
|
||||
)
|
||||
messages.append(
|
||||
{
|
||||
"role": "user",
|
||||
"content": json.dumps(
|
||||
parsed_request.model_dump(mode="json"),
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
),
|
||||
}
|
||||
)
|
||||
raw = self._generator(messages) if self._generator else self._generate(messages)
|
||||
try:
|
||||
payload = _extract_json_object(raw)
|
||||
return FinancialReasoningResponse.model_validate(payload)
|
||||
except (ValueError, json.JSONDecodeError, ValidationError):
|
||||
return FinancialReasoningResponse(
|
||||
status="insufficient_evidence",
|
||||
thesis="The model output could not be validated against the BLUM Finance schema.",
|
||||
confidence=0,
|
||||
what_would_change_the_view=[
|
||||
"Provide a schema-valid response grounded in the supplied evidence."
|
||||
],
|
||||
)
|
||||
|
||||
def _generate(self, messages: list[dict[str, str]]) -> str:
|
||||
if self.runtime == "mlx":
|
||||
return self._generate_with_mlx(messages)
|
||||
return self._generate_with_transformers(messages)
|
||||
|
||||
def _generate_with_transformers(self, messages: list[dict[str, str]]) -> str:
|
||||
if self._pipeline is None:
|
||||
from transformers import pipeline
|
||||
|
||||
self._pipeline = pipeline(
|
||||
"text-generation",
|
||||
model=self.model_id,
|
||||
revision=self.revision,
|
||||
device_map="auto",
|
||||
)
|
||||
result = self._pipeline(
|
||||
messages,
|
||||
max_new_tokens=768,
|
||||
do_sample=False,
|
||||
return_full_text=False,
|
||||
)
|
||||
generated = result[0]["generated_text"]
|
||||
if isinstance(generated, list):
|
||||
generated = generated[-1]["content"]
|
||||
return str(generated)
|
||||
|
||||
def _generate_with_mlx(self, messages: list[dict[str, str]]) -> str:
|
||||
try:
|
||||
from mlx_lm import generate, load
|
||||
from mlx_lm.sample_utils import make_sampler
|
||||
except ImportError as exc:
|
||||
raise RuntimeError(
|
||||
"MLX inference requires the 'mlx' optional dependencies on Apple Silicon."
|
||||
) from exc
|
||||
if self._mlx_model is None or self._mlx_tokenizer is None:
|
||||
self._mlx_model, self._mlx_tokenizer = load(
|
||||
self.model_id,
|
||||
revision=self.revision,
|
||||
tokenizer_config={"trust_remote_code": True},
|
||||
)
|
||||
prompt = self._mlx_tokenizer.apply_chat_template(
|
||||
messages,
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
enable_thinking=False,
|
||||
)
|
||||
return str(
|
||||
generate(
|
||||
self._mlx_model,
|
||||
self._mlx_tokenizer,
|
||||
prompt=prompt,
|
||||
max_tokens=768,
|
||||
sampler=make_sampler(temp=0.0),
|
||||
verbose=False,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def _extract_json_object(text: str) -> dict:
|
||||
stripped = text.strip()
|
||||
if stripped.startswith("```"):
|
||||
stripped = stripped.removeprefix("```json").removeprefix("```")
|
||||
stripped = stripped.removesuffix("```").strip()
|
||||
start = stripped.find("{")
|
||||
end = stripped.rfind("}")
|
||||
if start < 0 or end <= start:
|
||||
raise ValueError("No JSON object found.")
|
||||
return json.loads(stripped[start : end + 1])
|
||||
223
blum_finance/memory.py
Normal file
223
blum_finance/memory.py
Normal file
@@ -0,0 +1,223 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
import argparse
|
||||
from datetime import datetime, timezone
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
from .schemas import FinancialReasoningRequest, FinancialReasoningResponse
|
||||
|
||||
|
||||
TOKEN_PATTERN = re.compile(r"[A-Za-z0-9_]{2,}")
|
||||
|
||||
|
||||
class InvalidMemoryRecord(ValueError):
|
||||
"""Raised when a record cannot safely become retrieval memory."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StoredMemoryRecord:
|
||||
content_hash: str
|
||||
payload: dict[str, Any]
|
||||
|
||||
|
||||
class BlumFinanceMemoryStore:
|
||||
"""Small, auditable local memory with strict point-in-time retrieval.
|
||||
|
||||
This store never changes model weights. It exposes only matured observations
|
||||
available before a new request and labels them as historical analogies.
|
||||
"""
|
||||
|
||||
def __init__(self, path: str | Path) -> None:
|
||||
self.path = Path(path).expanduser().resolve()
|
||||
|
||||
def add(self, payload: dict[str, Any]) -> StoredMemoryRecord:
|
||||
normalized = _validate_memory_payload(payload)
|
||||
content_hash = _content_hash(normalized)
|
||||
row = {"content_hash": content_hash, "payload": normalized}
|
||||
existing = self._rows()
|
||||
if not any(item.get("content_hash") == content_hash for item in existing):
|
||||
self._replace([*existing, row])
|
||||
return StoredMemoryRecord(content_hash=content_hash, payload=normalized)
|
||||
|
||||
def add_bundle(self, bundle: str | Path | dict[str, Any]) -> StoredMemoryRecord:
|
||||
from .contributions import validate_contribution_bundle
|
||||
|
||||
if isinstance(bundle, (str, Path)):
|
||||
value = json.loads(Path(bundle).read_text(encoding="utf-8"))
|
||||
else:
|
||||
value = bundle
|
||||
validation = validate_contribution_bundle(value)
|
||||
if not validation.accepted:
|
||||
raise InvalidMemoryRecord(
|
||||
"Contribution is not eligible for memory: " + ", ".join(validation.blockers)
|
||||
)
|
||||
return self.add(value["payload"])
|
||||
|
||||
def retrieve(
|
||||
self,
|
||||
request: FinancialReasoningRequest,
|
||||
*,
|
||||
limit: int = 3,
|
||||
) -> list[dict[str, Any]]:
|
||||
if limit <= 0:
|
||||
return []
|
||||
request_tokens = _request_tokens(request.model_dump(mode="json"))
|
||||
candidates: list[tuple[float, datetime, dict[str, Any]]] = []
|
||||
for row in self._rows():
|
||||
payload = row.get("payload")
|
||||
if not isinstance(payload, dict):
|
||||
continue
|
||||
try:
|
||||
normalized = _validate_memory_payload(payload)
|
||||
observed_at = _timestamp(normalized["outcome"]["observed_at"])
|
||||
except (InvalidMemoryRecord, KeyError, TypeError, ValueError):
|
||||
continue
|
||||
if observed_at > _aware(request.as_of):
|
||||
continue
|
||||
memory_request = normalized["request"]
|
||||
memory_tokens = _request_tokens(memory_request)
|
||||
overlap = len(request_tokens & memory_tokens) / max(1, len(request_tokens | memory_tokens))
|
||||
ticker_match = str(memory_request.get("ticker", "")).upper() == request.ticker.upper()
|
||||
horizon_match = str(memory_request.get("horizon", "")) == request.horizon
|
||||
score = overlap + (2.0 if ticker_match else 0.0) + (0.5 if horizon_match else 0.0)
|
||||
candidates.append((score, observed_at, _retrieval_payload(normalized, row.get("content_hash"))))
|
||||
candidates.sort(key=lambda item: (item[0], item[1]), reverse=True)
|
||||
return [item[2] for item in candidates[:limit] if item[0] > 0]
|
||||
|
||||
def _rows(self) -> list[dict[str, Any]]:
|
||||
if not self.path.is_file():
|
||||
return []
|
||||
rows: list[dict[str, Any]] = []
|
||||
for line in self.path.read_text(encoding="utf-8").splitlines():
|
||||
try:
|
||||
value = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if isinstance(value, dict):
|
||||
rows.append(value)
|
||||
return rows
|
||||
|
||||
def _replace(self, rows: list[dict[str, Any]]) -> None:
|
||||
self.path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = self.path.with_suffix(self.path.suffix + ".tmp")
|
||||
body = "".join(_canonical_json(row) + "\n" for row in rows)
|
||||
temporary.write_text(body, encoding="utf-8")
|
||||
os.replace(temporary, self.path)
|
||||
|
||||
|
||||
def _validate_memory_payload(payload: dict[str, Any]) -> dict[str, Any]:
|
||||
if not isinstance(payload, dict):
|
||||
raise InvalidMemoryRecord("Memory payload must be an object")
|
||||
request = payload.get("request")
|
||||
response = payload.get("response")
|
||||
outcome = payload.get("outcome")
|
||||
quality = payload.get("quality")
|
||||
try:
|
||||
parsed_request = FinancialReasoningRequest.model_validate(request)
|
||||
parsed_response = FinancialReasoningResponse.model_validate(response)
|
||||
except Exception as exc:
|
||||
raise InvalidMemoryRecord(f"Invalid BLUM request or response: {exc}") from exc
|
||||
if not isinstance(outcome, dict) or not outcome.get("observed_at"):
|
||||
raise InvalidMemoryRecord("A matured outcome with observed_at is required")
|
||||
observed_at = _timestamp(outcome["observed_at"])
|
||||
if observed_at <= _aware(parsed_request.as_of):
|
||||
raise InvalidMemoryRecord("The outcome must be observed after the decision")
|
||||
status = str(outcome.get("status") or "").strip().lower()
|
||||
if status in {"", "pending", "unresolved", "inconclusive"}:
|
||||
raise InvalidMemoryRecord("The outcome is not mature")
|
||||
if not isinstance(quality, dict) or quality.get("source_verified") is not True:
|
||||
raise InvalidMemoryRecord("Memory requires verified source provenance")
|
||||
try:
|
||||
score = float(quality.get("score"))
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise InvalidMemoryRecord("Memory quality score is missing") from exc
|
||||
if score < 70:
|
||||
raise InvalidMemoryRecord("Memory quality is below 70")
|
||||
return {
|
||||
**payload,
|
||||
"request": parsed_request.model_dump(mode="json"),
|
||||
"response": parsed_response.model_dump(mode="json"),
|
||||
"outcome": dict(outcome),
|
||||
"quality": {**quality, "score": score},
|
||||
}
|
||||
|
||||
|
||||
def _retrieval_payload(payload: dict[str, Any], content_hash: Any) -> dict[str, Any]:
|
||||
request = payload["request"]
|
||||
response = payload["response"]
|
||||
outcome = payload["outcome"]
|
||||
return {
|
||||
"memory_id": str(content_hash or _content_hash(payload)),
|
||||
"ticker": request.get("ticker"),
|
||||
"horizon": request.get("horizon"),
|
||||
"decision_as_of": request.get("as_of"),
|
||||
"observed_at": outcome.get("observed_at"),
|
||||
"prior_status": response.get("status"),
|
||||
"prior_thesis": response.get("thesis"),
|
||||
"outcome": {
|
||||
key: outcome.get(key)
|
||||
for key in ("status", "realized_r", "benchmark_excess")
|
||||
if outcome.get(key) is not None
|
||||
},
|
||||
"lesson": outcome.get("lesson") or "No explicit lesson was supplied.",
|
||||
"quality_score": payload["quality"]["score"],
|
||||
}
|
||||
|
||||
|
||||
def _request_tokens(payload: dict[str, Any]) -> set[str]:
|
||||
return {
|
||||
token.lower()
|
||||
for token in TOKEN_PATTERN.findall(json.dumps(payload, ensure_ascii=False, sort_keys=True))
|
||||
}
|
||||
|
||||
|
||||
def _content_hash(payload: dict[str, Any]) -> str:
|
||||
return hashlib.sha256(_canonical_json(payload).encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def _canonical_json(payload: Any) -> str:
|
||||
return json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
||||
|
||||
|
||||
def _timestamp(value: Any) -> datetime:
|
||||
if isinstance(value, datetime):
|
||||
return _aware(value)
|
||||
parsed = datetime.fromisoformat(str(value).replace("Z", "+00:00"))
|
||||
return _aware(parsed)
|
||||
|
||||
|
||||
def _aware(value: datetime) -> datetime:
|
||||
if value.tzinfo is None:
|
||||
return value.replace(tzinfo=timezone.utc)
|
||||
return value.astimezone(timezone.utc)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Import a validated BLUM contribution into local point-in-time memory."
|
||||
)
|
||||
parser.add_argument("bundle", type=Path)
|
||||
parser.add_argument(
|
||||
"--memory",
|
||||
type=Path,
|
||||
default=Path.home() / ".blum-finance" / "memory.jsonl",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
stored = BlumFinanceMemoryStore(args.memory).add_bundle(args.bundle)
|
||||
print(
|
||||
json.dumps(
|
||||
{"status": "stored", "content_hash": stored.content_hash, "memory": str(args.memory)},
|
||||
indent=2,
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
64
blum_finance/schemas.py
Normal file
64
blum_finance/schemas.py
Normal file
@@ -0,0 +1,64 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Any, Literal
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
||||
|
||||
|
||||
ReasoningStatus = Literal[
|
||||
"avoid",
|
||||
"watch",
|
||||
"wait_for_trigger",
|
||||
"actionable_if_confirmed",
|
||||
"manage_open_position",
|
||||
"reduce",
|
||||
"exit",
|
||||
"insufficient_evidence",
|
||||
]
|
||||
|
||||
|
||||
class EvidenceItem(BaseModel):
|
||||
model_config = ConfigDict(extra="allow")
|
||||
|
||||
type: str = Field(min_length=1)
|
||||
value: Any
|
||||
source: str | None = None
|
||||
observed_at: datetime | None = None
|
||||
|
||||
|
||||
class FinancialReasoningRequest(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
|
||||
ticker: str = Field(min_length=1, max_length=32)
|
||||
as_of: datetime
|
||||
horizon: str = "swing"
|
||||
market_context: dict[str, Any] = Field(default_factory=dict)
|
||||
portfolio_context: dict[str, Any] = Field(default_factory=dict)
|
||||
evidence: list[EvidenceItem] = Field(min_length=1)
|
||||
question: str = "Evaluate the evidence and state what would change the view."
|
||||
|
||||
|
||||
class FinancialReasoningResponse(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
|
||||
status: ReasoningStatus
|
||||
thesis: str = Field(min_length=1)
|
||||
bull_case: list[str] = Field(default_factory=list)
|
||||
bear_case: list[str] = Field(default_factory=list)
|
||||
risks: list[str] = Field(default_factory=list)
|
||||
invalidation_conditions: list[str] = Field(default_factory=list)
|
||||
confidence: float = Field(ge=0, le=100)
|
||||
what_would_change_the_view: list[str] = Field(min_length=1)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def require_risk_definition_for_actionable_states(self) -> "FinancialReasoningResponse":
|
||||
actionable = {
|
||||
"actionable_if_confirmed",
|
||||
"manage_open_position",
|
||||
"reduce",
|
||||
"exit",
|
||||
}
|
||||
if self.status in actionable and (not self.risks or not self.invalidation_conditions):
|
||||
raise ValueError("Actionable states require risks and invalidation conditions.")
|
||||
return self
|
||||
Reference in New Issue
Block a user