516 lines
19 KiB
Python
516 lines
19 KiB
Python
"""Journal-adapter editorial policy. Not a general provenance or privacy rule.
|
|
|
|
Fact fidelity is not wording fidelity. Mixed prose, notes and fragments in the
|
|
same day are the normal case and are handled in one generate call. Historical
|
|
texts are style references, never today's facts.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import re
|
|
from difflib import SequenceMatcher
|
|
|
|
from journal_body import plain_text
|
|
from journal_generation_policy import (
|
|
EMPTY_STYLE_CONTEXT,
|
|
STYLE_CONTEXT_KEYS,
|
|
normalize_style_context,
|
|
)
|
|
from writing_profile_schema import is_meta_style_text, recency_weight
|
|
from writing_profile_store import (
|
|
clip_field,
|
|
get_profile,
|
|
has_confirmed_profile,
|
|
list_style_sources,
|
|
)
|
|
|
|
STYLE_EXAMPLE_MAX = 2
|
|
STYLE_EXAMPLE_CHARS = 900
|
|
MIN_EXAMPLE_CHARS = 40
|
|
|
|
GENERATE_SEED_REVISION = "2026-08-29-voice-legacy-immutable-v1"
|
|
|
|
EMPTY_STYLE_EXAMPLES = (
|
|
"Keine historischen Stilbeispiele. Es gilt nur WRITING_PROFILE "
|
|
"(oder der neutrale Journalstil)."
|
|
)
|
|
|
|
|
|
def lexical_similarity(left: str, right: str) -> float:
|
|
"""Diagnostic only. Must not reject a draft or trigger a retry."""
|
|
a = re.sub(r"\s+", " ", (left or "").strip().lower())
|
|
b = re.sub(r"\s+", " ", (right or "").strip().lower())
|
|
if not a or not b:
|
|
return 0.0
|
|
return round(SequenceMatcher(None, a, b).ratio(), 3)
|
|
|
|
|
|
_DANGLING_DETERMINER = re.compile(
|
|
r"(?i)\b(?:der|die|das|des|dem|den|ein|eine|einem|einen|einer)\s*$"
|
|
)
|
|
_DETERMINER_BEFORE_FINITE = re.compile(
|
|
r"(?i)\b(?:des|dem|den|der|die|das|ein|eine|einem|einen|einer)\s+"
|
|
r"(?:kann|können|konnte|muss|müssen|will|wollen|soll|sollen|"
|
|
r"ist|sind|war|waren|wird|werden|hat|haben|hatte|"
|
|
r"geht|gehen|ging|kam|kommen|kommt|gelangen|gelangte)\b"
|
|
)
|
|
|
|
|
|
def incomplete_syntax_markers(text: str) -> int:
|
|
"""Diagnostic count of dangling determiners or unpunctuated long clauses.
|
|
|
|
Does not rewrite text and must not trigger a retry.
|
|
"""
|
|
body = (text or "").strip()
|
|
if not body:
|
|
return 0
|
|
count = 0
|
|
clauses = [part.strip() for part in re.split(r"(?<=[.!?])\s+|\n+", body) if part.strip()]
|
|
for clause in clauses:
|
|
bare = clause.rstrip(".!?…\"»'")
|
|
if _DANGLING_DETERMINER.search(bare):
|
|
count += 1
|
|
if _DETERMINER_BEFORE_FINITE.search(clause):
|
|
count += 1
|
|
for para in re.split(r"\n\s*\n", body):
|
|
chunk = para.strip()
|
|
if len(chunk.split()) >= 8 and not re.search(r"[.!?]", chunk):
|
|
count += 1
|
|
return count
|
|
|
|
|
|
def style_example_diagnostics(examples: list[dict]) -> dict:
|
|
return {
|
|
"count": len(examples or []),
|
|
"kinds": [item.get("kind") or "style" for item in (examples or [])],
|
|
"chars": sum(len(item.get("excerpt") or "") for item in (examples or [])),
|
|
}
|
|
|
|
|
|
def writing_profile_trace(
|
|
profile_id: str,
|
|
*,
|
|
style_context: dict | None = None,
|
|
brief: str | None = None,
|
|
trait_count: int | None = None,
|
|
) -> dict:
|
|
"""Presence metadata only. No profile text, no labels.
|
|
|
|
When style_context or a compiled brief is supplied, the flags describe the
|
|
effective provider input rather than the stored profile inventory.
|
|
"""
|
|
from writing_profile_store import (
|
|
NEUTRAL_JOURNAL_STYLE,
|
|
compile_task_brief,
|
|
get_profile,
|
|
has_confirmed_profile,
|
|
)
|
|
|
|
confirmed = has_confirmed_profile(profile_id)
|
|
profile = get_profile(profile_id)
|
|
context = normalize_style_context(style_context, slot="voice") if style_context is not None else None
|
|
core = ((profile.get("core") or {}).get("value") or "").strip()
|
|
facet = next(
|
|
(
|
|
item
|
|
for item in profile.get("facets") or []
|
|
if item.get("facet_key") == "autobiographical_journal" and (item.get("value") or "").strip()
|
|
),
|
|
None,
|
|
)
|
|
traits = [
|
|
item
|
|
for item in profile.get("traits") or []
|
|
if item.get("status") == "active" and (item.get("statement") or "").strip()
|
|
]
|
|
if brief is None:
|
|
brief = compile_task_brief(
|
|
profile_id,
|
|
"journal_generate",
|
|
include_core=True if context is None else context["include_core"],
|
|
include_facet=True if context is None else context["include_facet"],
|
|
include_traits=True if context is None else context["include_traits"],
|
|
)
|
|
has_core = "Core:" in (brief or "")
|
|
has_facet = "Facet-Delta:" in (brief or "") or "Autobiografisches Journaling" in (brief or "")
|
|
if trait_count is None:
|
|
trait_count = sum(1 for line in (brief or "").splitlines() if line.startswith("- "))
|
|
present = bool(brief and brief != NEUTRAL_JOURNAL_STYLE)
|
|
return {
|
|
"confirmed": confirmed,
|
|
"present": present,
|
|
"neutral_fallback": brief == NEUTRAL_JOURNAL_STYLE,
|
|
"has_core": has_core,
|
|
"has_facet": has_facet,
|
|
"trait_count": int(trait_count or 0),
|
|
"brief_chars": len(brief or ""),
|
|
"inventory_has_core": bool(confirmed and core),
|
|
"inventory_has_facet": bool(confirmed and facet),
|
|
"inventory_trait_count": len(traits) if confirmed else 0,
|
|
}
|
|
|
|
|
|
def narration_sources_text(artifact: dict) -> str:
|
|
"""Present attested day facts to the model. Not a wording template, not JSON."""
|
|
parts: list[str] = []
|
|
for item in artifact.get("sources") or []:
|
|
if not isinstance(item, dict):
|
|
continue
|
|
text = (item.get("text") or "").strip()
|
|
if not text:
|
|
continue
|
|
source_id = (item.get("source_id") or "").strip()
|
|
prefix = f"[{source_id}]\n" if source_id else ""
|
|
parts.append(prefix + text)
|
|
if parts:
|
|
return "\n\n".join(parts)
|
|
from journal_reconstruct import claim_texts
|
|
|
|
fallback = [str(part).strip() for part in claim_texts(artifact) if str(part).strip()]
|
|
return "\n\n".join(fallback)
|
|
|
|
|
|
def format_style_examples(examples: list[dict]) -> str:
|
|
if not examples:
|
|
return EMPTY_STYLE_EXAMPLES
|
|
lines = [
|
|
"Nur Ton, Rhythmus und sprachliche Entscheidungen. "
|
|
"Ereignisse, Personen, Orte und Bewertungen aus diesen Beispielen "
|
|
"sind keine Tatsachen des heutigen Eintrags und dürfen nicht übernommen werden."
|
|
]
|
|
for index, item in enumerate(examples, start=1):
|
|
kind = item.get("kind") or "style"
|
|
when = (item.get("occurred_at") or "")[:10]
|
|
header = f"Beispiel {index} ({kind}" + (f", {when}" if when else "") + "):"
|
|
lines.append(header)
|
|
lines.append((item.get("excerpt") or "").strip())
|
|
return "\n".join(part for part in lines if part).strip()
|
|
|
|
|
|
def wrap_style_block(title: str, body: str) -> str:
|
|
text = (body or "").strip()
|
|
if not text:
|
|
return ""
|
|
return f"{title}\n{text}\n"
|
|
|
|
|
|
def _estimated_tokens(text: str) -> int:
|
|
from prompt_budget import estimate_tokens
|
|
|
|
body = text or ""
|
|
if not body:
|
|
return 0
|
|
return estimate_tokens(body)
|
|
|
|
|
|
def compile_journal_style_application(
|
|
profile_id: str,
|
|
style_context: dict | None,
|
|
*,
|
|
exclude_dates: list[str] | None = None,
|
|
exclude_entry_ids: list[str] | None = None,
|
|
) -> dict:
|
|
"""Select only the style sources released by the voice guideline.
|
|
|
|
Returns prompt fragments and an effective trace. Empty sections stay empty
|
|
so the prompt builder can omit their headings.
|
|
"""
|
|
from writing_profile_store import (
|
|
NEUTRAL_JOURNAL_STYLE,
|
|
compile_task_brief,
|
|
has_confirmed_profile,
|
|
)
|
|
|
|
context = normalize_style_context(style_context or EMPTY_STYLE_CONTEXT, slot="voice")
|
|
confirmed = has_confirmed_profile(profile_id)
|
|
omitted: list[dict] = []
|
|
requested = {key: bool(context.get(key)) for key in STYLE_CONTEXT_KEYS}
|
|
|
|
def omit(part: str, reason: str) -> None:
|
|
omitted.append({"part": part, "reason": reason})
|
|
|
|
if not requested["include_core"]:
|
|
omit("core", "not_selected")
|
|
if not requested["include_facet"]:
|
|
omit("facet", "not_selected")
|
|
if not requested["include_traits"]:
|
|
omit("traits", "not_selected")
|
|
if not requested["include_style_examples"]:
|
|
omit("style_examples", "not_selected")
|
|
|
|
wants_profile = any(requested[key] for key in ("include_core", "include_facet", "include_traits"))
|
|
brief = compile_task_brief(
|
|
profile_id,
|
|
"journal_generate",
|
|
include_core=requested["include_core"],
|
|
include_facet=requested["include_facet"],
|
|
include_traits=requested["include_traits"],
|
|
)
|
|
if wants_profile and not confirmed:
|
|
for part, flag in (
|
|
("core", "include_core"),
|
|
("facet", "include_facet"),
|
|
("traits", "include_traits"),
|
|
):
|
|
if requested[flag] and not any(item["part"] == part and item["reason"] == "unconfirmed" for item in omitted):
|
|
omit(part, "unconfirmed")
|
|
elif confirmed:
|
|
if requested["include_core"] and "Core:" not in (brief or ""):
|
|
omit("core", "empty")
|
|
if requested["include_facet"] and "Facet-Delta:" not in (brief or "") and "Autobiografisches Journaling" not in (brief or ""):
|
|
omit("facet", "empty")
|
|
if requested["include_traits"] and not any(line.startswith("- ") for line in (brief or "").splitlines()):
|
|
omit("traits", "empty")
|
|
|
|
example_rows: list[dict] = []
|
|
examples_text = ""
|
|
if requested["include_style_examples"]:
|
|
example_rows = select_journal_style_examples(
|
|
profile_id,
|
|
exclude_dates=exclude_dates,
|
|
exclude_entry_ids=exclude_entry_ids,
|
|
)
|
|
if example_rows:
|
|
examples_text = format_style_examples(example_rows)
|
|
else:
|
|
omit("style_examples", "empty")
|
|
|
|
profile_block = wrap_style_block("WRITING_PROFILE", brief)
|
|
examples_block = wrap_style_block("STYLE_EXAMPLES", examples_text)
|
|
trait_count = sum(1 for line in (brief or "").splitlines() if line.startswith("- "))
|
|
core_included = "Core:" in (brief or "")
|
|
facet_included = "Facet-Delta:" in (brief or "") or "Autobiografisches Journaling" in (brief or "")
|
|
return {
|
|
"style_context": context,
|
|
"brief": brief,
|
|
"profile_block": profile_block,
|
|
"style_examples": examples_text,
|
|
"examples_block": examples_block,
|
|
"example_rows": example_rows,
|
|
"omitted": omitted,
|
|
"requested": requested,
|
|
"effective": {
|
|
"include_core": core_included,
|
|
"include_facet": facet_included,
|
|
"include_traits": trait_count > 0,
|
|
"include_style_examples": bool(example_rows),
|
|
"trait_count": trait_count,
|
|
"style_example_count": len(example_rows),
|
|
"neutral_fallback": brief == NEUTRAL_JOURNAL_STYLE,
|
|
"blocks": [
|
|
{
|
|
"key": "writing_profile",
|
|
"included": bool(profile_block),
|
|
"chars": len(profile_block),
|
|
"estimated_tokens": _estimated_tokens(profile_block),
|
|
},
|
|
{
|
|
"key": "core",
|
|
"included": core_included,
|
|
"chars": len(brief.split("Autobiografisches Journaling")[0]) if core_included else 0,
|
|
"estimated_tokens": _estimated_tokens(brief) if core_included and not facet_included and trait_count == 0 else 0,
|
|
},
|
|
{
|
|
"key": "facet",
|
|
"included": facet_included,
|
|
"chars": 0,
|
|
"estimated_tokens": 0,
|
|
},
|
|
{
|
|
"key": "traits",
|
|
"included": trait_count > 0,
|
|
"count": trait_count,
|
|
"chars": 0,
|
|
"estimated_tokens": 0,
|
|
},
|
|
{
|
|
"key": "style_examples",
|
|
"included": bool(examples_block),
|
|
"count": len(example_rows),
|
|
"chars": len(examples_block),
|
|
"estimated_tokens": _estimated_tokens(examples_block),
|
|
},
|
|
],
|
|
},
|
|
}
|
|
|
|
|
|
def style_application_trace(
|
|
compiled_policy,
|
|
bundle: dict,
|
|
*,
|
|
dropped: list[str] | None = None,
|
|
) -> dict:
|
|
dropped = list(dropped or [])
|
|
omitted = [dict(item) for item in bundle.get("omitted") or []]
|
|
effective = dict(bundle.get("effective") or {})
|
|
example_rows = list(bundle.get("example_rows") or [])
|
|
profile_block = bundle.get("profile_block") or ""
|
|
examples_block = bundle.get("examples_block") or ""
|
|
if "style_examples" in dropped:
|
|
examples_block = ""
|
|
example_rows = []
|
|
effective["include_style_examples"] = False
|
|
effective["style_example_count"] = 0
|
|
if not any(item.get("part") == "style_examples" and item.get("reason") == "budget" for item in omitted):
|
|
omitted.append({"part": "style_examples", "reason": "budget"})
|
|
for block in effective.get("blocks") or []:
|
|
if block.get("key") == "style_examples":
|
|
block["included"] = False
|
|
block["count"] = 0
|
|
block["chars"] = 0
|
|
block["estimated_tokens"] = 0
|
|
brief = bundle.get("brief") or ""
|
|
core_chars = 0
|
|
facet_chars = 0
|
|
trait_chars = 0
|
|
if "Core:" in brief:
|
|
core_part = brief.split("Autobiografisches Journaling")[0]
|
|
core_chars = len(core_part.strip())
|
|
if "Autobiografisches Journaling" in brief:
|
|
rest = brief.split("Autobiografisches Journaling", 1)[1]
|
|
facet_part = rest.split("\n- ", 1)[0]
|
|
facet_chars = len(("Autobiografisches Journaling" + facet_part).strip())
|
|
trait_lines = [line for line in brief.splitlines() if line.startswith("- ")]
|
|
trait_chars = sum(len(line) for line in trait_lines)
|
|
blocks = [
|
|
{
|
|
"key": "writing_profile",
|
|
"included": bool(profile_block.strip()),
|
|
"chars": len(profile_block),
|
|
"estimated_tokens": _estimated_tokens(profile_block),
|
|
},
|
|
{
|
|
"key": "core",
|
|
"included": "Core:" in brief,
|
|
"chars": core_chars,
|
|
"estimated_tokens": _estimated_tokens(brief.split("\n", 1)[0] if "Core:" in brief else ""),
|
|
},
|
|
{
|
|
"key": "facet",
|
|
"included": "Autobiografisches Journaling" in brief,
|
|
"chars": facet_chars,
|
|
"estimated_tokens": _estimated_tokens("Autobiografisches Journaling" + brief.split("Autobiografisches Journaling", 1)[1].split("\n- ", 1)[0] if "Autobiografisches Journaling" in brief else ""),
|
|
},
|
|
{
|
|
"key": "traits",
|
|
"included": bool(trait_lines),
|
|
"count": len(trait_lines),
|
|
"chars": trait_chars,
|
|
"estimated_tokens": _estimated_tokens("\n".join(trait_lines)),
|
|
},
|
|
{
|
|
"key": "style_examples",
|
|
"included": bool(examples_block.strip()),
|
|
"count": len(example_rows),
|
|
"chars": len(examples_block),
|
|
"estimated_tokens": _estimated_tokens(examples_block),
|
|
},
|
|
]
|
|
requested = bundle.get("requested") or dict(EMPTY_STYLE_CONTEXT)
|
|
return {
|
|
"id": compiled_policy.ids.get("voice") if compiled_policy else "",
|
|
"key": compiled_policy.keys.get("voice") if compiled_policy else "",
|
|
"label": compiled_policy.labels.get("voice") if compiled_policy else "",
|
|
"revision": compiled_policy.revisions.get("voice") if compiled_policy else 1,
|
|
"cloned_from": (compiled_policy.cloned_from.get("voice") if compiled_policy else "") or "",
|
|
"requested": requested,
|
|
"effective": {
|
|
"include_core": any(block["key"] == "core" and block["included"] for block in blocks),
|
|
"include_facet": any(block["key"] == "facet" and block["included"] for block in blocks),
|
|
"trait_count": len(trait_lines),
|
|
"style_example_count": len(example_rows),
|
|
"neutral_fallback": bool((bundle.get("effective") or {}).get("neutral_fallback")),
|
|
"blocks": blocks,
|
|
},
|
|
"omitted": omitted,
|
|
"dropped_optional_blocks": dropped,
|
|
}
|
|
|
|
|
|
def _content_digest(text: str) -> str:
|
|
body = re.sub(r"\s+", " ", plain_text(text or "")).strip().lower()
|
|
return hashlib.sha256(body.encode("utf-8")).hexdigest()
|
|
|
|
|
|
def _eligible_excerpt(body: str, limit: int) -> str:
|
|
text = plain_text(body or "").strip()
|
|
if len(text) < MIN_EXAMPLE_CHARS or is_meta_style_text(text):
|
|
return ""
|
|
return clip_field(text, limit)
|
|
|
|
|
|
def select_journal_style_examples(
|
|
profile_id: str,
|
|
*,
|
|
exclude_dates: list[str] | None = None,
|
|
exclude_entry_ids: list[str] | None = None,
|
|
max_n: int = STYLE_EXAMPLE_MAX,
|
|
max_chars: int = STYLE_EXAMPLE_CHARS,
|
|
) -> list[dict]:
|
|
"""Final accepted journal texts first, then imports, then trait excerpts.
|
|
|
|
Current-day sources are excluded. Dialogue is never a style authority.
|
|
"""
|
|
excluded_dates = {(item or "")[:10] for item in (exclude_dates or []) if item}
|
|
excluded_ids = {item for item in (exclude_entry_ids or []) if item}
|
|
ranked = list_style_sources(profile_id)
|
|
buckets = [
|
|
("journal_entry", ranked.get("journal_entry") or []),
|
|
("imported_text", ranked.get("imported_text") or []),
|
|
]
|
|
picked: list[dict] = []
|
|
seen: set[str] = set()
|
|
|
|
def consider(kind: str, item: dict, excerpt: str) -> None:
|
|
if len(picked) >= max_n or not excerpt:
|
|
return
|
|
digest = _content_digest(excerpt)
|
|
if digest in seen:
|
|
return
|
|
seen.add(digest)
|
|
picked.append(
|
|
{
|
|
"kind": kind,
|
|
"excerpt": excerpt,
|
|
"occurred_at": item.get("occurred_at") or item.get("created"),
|
|
"entry_id": item.get("entry_id"),
|
|
"weight": item.get("weight") or recency_weight(item.get("occurred_at")),
|
|
}
|
|
)
|
|
|
|
for kind, rows in buckets:
|
|
ordered = sorted(
|
|
rows,
|
|
key=lambda row: (
|
|
-float(row.get("weight") or 0),
|
|
-recency_weight(row.get("occurred_at") or row.get("created")),
|
|
),
|
|
)
|
|
for item in ordered:
|
|
if len(picked) >= max_n:
|
|
return picked
|
|
entry_id = (item.get("entry_id") or "").strip()
|
|
if entry_id and entry_id in excluded_ids:
|
|
continue
|
|
when = (item.get("occurred_at") or "")[:10]
|
|
if when and when in excluded_dates:
|
|
continue
|
|
consider(kind, item, _eligible_excerpt(item.get("body") or "", max_chars))
|
|
|
|
if picked:
|
|
return picked
|
|
|
|
profile = get_profile(profile_id)
|
|
if not has_confirmed_profile(profile_id):
|
|
return picked
|
|
for trait in profile.get("traits") or []:
|
|
if len(picked) >= max_n:
|
|
break
|
|
for ref in trait.get("exemplars") or []:
|
|
excerpt = _eligible_excerpt(ref.get("excerpt") or "", max_chars)
|
|
consider("trait_exemplar", ref, excerpt)
|
|
if len(picked) >= max_n:
|
|
break
|
|
return picked
|