Repository object · implementation
ruff: noqa: E501
Accepted implementation in the public catalog.
- Source path
src/epistemedia/open_dockets.py- Media type
text/x-python- Object ID
em:implementation:sha256:07fc3391fdfdd119191e004d117cf1bdd288d85d3dae09feb883e266632144b3- Content digest
b0eb1657d28f0f3e6e6df27dee362e8e524ac993b80bfa26d1a80699ac4878a0
Also filed under
Source content
ruff: noqa: E501
"""GitHub-native, independently reviewed open-docket contributions."""
from __future__ import annotations
import datetime as dt
import hashlib
import html
import json
import math
import re
from dataclasses import dataclass
from pathlib import Path
from typing import Any
from .research_kit import SECRET_PATTERNS, parse_utc_timestamp, validate_proposal
INTAKE_FORMAT = "epistemedia-open-docket-intake-v0.2"
TRACE_FORMAT = "epistemedia-disclosure-safe-action-trace-v0.1"
REVIEW_FORMAT = "epistemedia-open-docket-review-v0.2"
CONTROLLER_ATTESTATION_FORMAT = "epistemedia-controller-runtime-attestation-v0.1"
DOCKET_FORMAT = "epistemedia-open-docket-v0.2"
PROMOTION_RECEIPT_FORMAT = "epistemedia-open-docket-promotion-receipt-v0.2"
SUBMISSION_ROOT = Path("research/open-dockets/submissions")
ACCEPTED_ROOT = Path("research/open-dockets")
DOSSIER_MANIFEST_ROOT = Path("catalog/dossiers")
QUESTION_STOPWORDS = {
"a",
"an",
"and",
"are",
"as",
"at",
"be",
"by",
"did",
"do",
"does",
"for",
"from",
"how",
"in",
"is",
"it",
"of",
"on",
"or",
"the",
"to",
"was",
"were",
"what",
"when",
"where",
"which",
"who",
"why",
"with",
}
SUBJECT_KEY_STOPWORDS = QUESTION_STOPWORDS | {"case", "claim", "research"}
CONCEPT_ALIASES = {
"agents": "agent",
"citations": "citation",
"cited": "citation",
"citing": "citation",
"claims": "claim",
"claimed": "claim",
"communications": "communication",
"corrected": "correction",
"correcting": "correction",
"corrections": "correction",
"percentiles": "percentile",
"ranked": "rank",
"ranking": "rank",
"ranks": "rank",
"reported": "report",
"reporting": "report",
"reports": "report",
"scored": "score",
"scores": "score",
"weighted": "weight",
"weighting": "weight",
"weights": "weight",
}
MAX_TRACE_EVENTS = 100
MAX_TRACE_COST_AMOUNT = 1_000_000
SHA256 = re.compile(r"^[0-9a-f]{64}$")
GIT_OBJECT_ID = re.compile(r"^(?:[0-9a-f]{40}|[0-9a-f]{64})$")
SLUG = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
MODEL_FAMILY = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
INTAKE_FIELDS = {
"format",
"status",
"proposal_id",
"proposal_sha256",
"proposal_bytes",
"pr_body_sha256",
"pr_body_bytes",
"submitted_at",
"submitter",
"trace",
"credit",
"queue",
}
EMAIL = re.compile(r"(?i)\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b")
TRACE_ACTIONS = {
"read-protocol": "public-protocol-read",
"retrieve-source": "source-payload-omitted",
"validate-proposal": "validation-result-recorded",
"prepare-submission": "submission-bundle-prepared",
}
TRACE_STATUSES = {"completed", "failed", "partial"}
TRACE_FAILURE_CODES = {
"source-retrieval-failed",
"source-access-blocked",
"source-identity-unresolved",
"span-not-located",
"calculation-not-reproduced",
"proposal-validation-failed",
"git-operation-failed",
"pr-creation-failed",
"unknown-failure",
}
TRACE_INTERVENTION_CODES = {
"owner-clarification",
"credential-provision",
"paid-provider-authorization",
"manual-source-retrieval",
"proposal-repair",
"git-repair",
}
TRACE_CURRENCIES = {"USD", "unknown"}
TRACE_COST_BASES = {
"provider-reported",
"subscription-no-marginal-cost",
"not-incurred",
"unknown",
}
def canonical_json(value: Any) -> bytes:
return (
json.dumps(value, indent=2, ensure_ascii=False, sort_keys=True) + "\n"
).encode("utf-8")
def sha256_bytes(value: bytes) -> str:
return hashlib.sha256(value).hexdigest()
def _concept_tokens(value: str) -> tuple[str, ...]:
normalized = value.casefold()
replacements = (
(r"\b(?:seven\s*[-–—/]\s*thirty[ -]?eight\s*[-–—/]\s*fifty[ -]?five|7\s*[-–—/]\s*38\s*[-–—/]\s*55)\b", " rule73855 "),
(r"\bgpt\s*[-–—]?\s*4\b", " gpt4 "),
(r"\buniform\s+bar\s+examination\b", " ube "),
(r"\b(?:simulated\s+)?bar\s+(?:exam(?:ination)?|test)\b", " ube "),
(r"\bninetieth\b", " 90th "),
(r"\bninety\b", " 90 "),
)
for pattern, replacement in replacements:
normalized = re.sub(pattern, replacement, normalized)
return tuple(
CONCEPT_ALIASES.get(token, token)
for token in re.findall(r"[a-z0-9]+", normalized)
if token not in QUESTION_STOPWORDS
)
def _closely_restates(question: str, accepted_question: str, subject_key: str) -> bool:
candidate_tokens = _concept_tokens(question)
accepted_tokens = _concept_tokens(accepted_question)
if not candidate_tokens or not accepted_tokens:
return question.strip().casefold() == accepted_question.strip().casefold()
if candidate_tokens == accepted_tokens:
return True
candidate = set(candidate_tokens)
accepted = set(accepted_tokens)
subject_anchors = set(_concept_tokens(subject_key)) - SUBJECT_KEY_STOPWORDS
shared_anchors = candidate & subject_anchors
if "rule73855" in shared_anchors:
return True
if len(shared_anchors) >= 2:
return True
if len(shared_anchors) == 1:
shared_context = (candidate & accepted) - shared_anchors
return len(shared_context) >= 2
return False
def _load_prior_art_record(
root: Path,
record_path: Path,
*,
label: str,
subject_key: str,
errors: list[str],
) -> tuple[str, str, str] | None:
if record_path.is_symlink() or not record_path.is_file():
errors.append(f"{label} record is missing or is not a regular file: {record_path.relative_to(root)}")
return None
try:
record = json.loads(record_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as exc:
errors.append(f"{label} record is unreadable or malformed: {exc}")
return None
if not isinstance(record, dict):
errors.append(f"{label} record must be a JSON object")
return None
question = record.get("question")
if not isinstance(question, str) or not question.strip():
errors.append(f"{label} record question must be a non-empty string")
return None
return label, subject_key, question
def validate_question_novelty(root: Path, question: str) -> list[str]:
"""Reject subject-aware near duplicates and fail closed on prior-art gaps."""
errors: list[str] = []
if not isinstance(question, str) or not question.strip():
return ["proposal question must be a non-empty string for prior-art comparison"]
resolved_root = root.resolve()
manifest_root = root / DOSSIER_MANIFEST_ROOT
if manifest_root.is_symlink() or not manifest_root.is_dir():
return [f"accepted dossier manifest directory is missing or invalid: {DOSSIER_MANIFEST_ROOT}"]
manifest_paths = sorted(manifest_root.glob("*.json"))
if not manifest_paths:
return [f"accepted dossier manifest directory contains no JSON records: {DOSSIER_MANIFEST_ROOT}"]
comparisons: list[tuple[str, str, str]] = []
for manifest_path in manifest_paths:
label = f"accepted dossier {manifest_path.stem}"
if manifest_path.is_symlink() or not manifest_path.is_file():
errors.append(f"{label} manifest is missing or is not a regular file")
continue
try:
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as exc:
errors.append(f"{label} manifest is unreadable or malformed: {exc}")
continue
if not isinstance(manifest, dict):
errors.append(f"{label} manifest must be a JSON object")
continue
dossier_path_value = manifest.get("dossier_path")
if not isinstance(dossier_path_value, str) or not dossier_path_value.strip():
errors.append(f"{label} manifest dossier_path must be a non-empty string")
continue
dossier_path = (root / dossier_path_value).resolve()
try:
dossier_path.relative_to(resolved_root)
except ValueError:
errors.append(f"{label} dossier_path escapes the repository root")
continue
record = _load_prior_art_record(
root,
dossier_path,
label=label,
subject_key=manifest_path.stem,
errors=errors,
)
if record is not None:
comparisons.append(record)
accepted_root = root / ACCEPTED_ROOT
if accepted_root.is_symlink():
errors.append(f"reviewed open-docket root is a symlink: {ACCEPTED_ROOT}")
elif accepted_root.exists():
if not accepted_root.is_dir():
errors.append(f"reviewed open-docket root is not a regular directory: {ACCEPTED_ROOT}")
else:
for docket_path in sorted(accepted_root.iterdir()):
if docket_path.name == "submissions":
continue
label = f"reviewed open docket {docket_path.name}"
if docket_path.is_symlink():
errors.append(f"{label} path must not be a symlink")
continue
if not docket_path.is_dir():
continue
record = _load_prior_art_record(
root,
docket_path / "proposal.json",
label=label,
subject_key=docket_path.name,
errors=errors,
)
if record is not None:
comparisons.append(record)
errors.extend(
f"proposal question closely restates {label}"
for label, subject_key, accepted_question in comparisons
if _closely_restates(question, accepted_question, subject_key)
)
return errors
def _contains_disallowed_text(value: Any) -> bool:
if isinstance(value, str):
lowered = value.lower()
if any(pattern.search(value) for pattern in SECRET_PATTERNS):
return True
if EMAIL.search(value) or re.search(r"(?<![A-Za-z0-9])/(?:tmp|var|opt|etc)/", value):
return True
if re.search(r"(?:^|[\s'\"`])(?:\.\.[/\\])+", value):
return True
if re.search(r"(?:^|[\s'\"`])[A-Za-z]:\\", value):
return True
return any(
marker in lowered
for marker in (
"chain of thought",
"chain-of-thought",
"private reasoning",
"hidden reasoning",
"system prompt",
"developer message",
)
)
if isinstance(value, list):
return any(_contains_disallowed_text(item) for item in value)
if isinstance(value, dict):
return any(
_contains_disallowed_text(key) or _contains_disallowed_text(item)
for key, item in value.items()
)
return False
def _normalized_toolchain(values: Any) -> set[str]:
if not isinstance(values, list):
return set()
return {
" ".join(re.sub(r"[^a-z0-9]+", " ", item.casefold()).split())
for item in values
if isinstance(item, str) and item.strip()
}
def validate_action_trace(trace: Any) -> list[str]:
errors: list[str] = []
if not isinstance(trace, dict):
return ["trace must be an object"]
expected = {"format", "events", "failures", "interventions", "cost"}
if set(trace) != expected:
errors.append("trace fields must exactly match the disclosure-safe trace format")
if trace.get("format") != TRACE_FORMAT:
errors.append(f"trace.format must equal {TRACE_FORMAT}")
events = trace.get("events")
if not isinstance(events, list) or not events:
errors.append("trace.events must be a non-empty list")
events = []
if len(events) > MAX_TRACE_EVENTS:
errors.append(f"trace.events exceeds {MAX_TRACE_EVENTS} items")
event_fields = {"sequence", "action", "target", "status", "artifact_sha256", "note"}
for index, event in enumerate(events):
path = f"trace.events[{index}]"
if not isinstance(event, dict) or set(event) != event_fields:
errors.append(f"{path} has unsupported or missing fields")
continue
if event.get("sequence") != index + 1:
errors.append(f"{path}.sequence must be contiguous from 1")
action = event.get("action")
if action not in TRACE_ACTIONS:
errors.append(f"{path}.action is not a supported trace action")
if event.get("status") not in TRACE_STATUSES:
errors.append(f"{path}.status is not a supported trace status")
if event.get("note") != TRACE_ACTIONS.get(action):
errors.append(f"{path}.note must use the fixed disclosure-safe action code")
for field in event_fields - {"sequence", "artifact_sha256"}:
if not isinstance(event.get(field), str) or not event[field].strip():
errors.append(f"{path}.{field} must be a non-empty string")
else:
limit = 2_048 if field == "target" else 80
if len(event[field]) > limit:
errors.append(f"{path}.{field} exceeds {limit} characters")
artifact = event.get("artifact_sha256")
if artifact != "none" and (not isinstance(artifact, str) or not SHA256.fullmatch(artifact)):
errors.append(f"{path}.artifact_sha256 must be a SHA-256 or none")
for key in ("failures", "interventions"):
if not isinstance(trace.get(key), list) or not all(
isinstance(item, str) and item.strip() for item in trace.get(key, [])
):
errors.append(f"trace.{key} must be a list of non-empty strings")
elif len(trace[key]) > 25:
errors.append(f"trace.{key} exceeds disclosure-safe item or size limits")
else:
allowed = TRACE_FAILURE_CODES if key == "failures" else TRACE_INTERVENTION_CODES
if any(item not in allowed for item in trace[key]):
errors.append(f"trace.{key} must contain only disclosure-safe codes")
cost = trace.get("cost")
if not isinstance(cost, dict) or set(cost) != {"amount", "currency", "basis"}:
errors.append("trace.cost must contain amount, currency, and basis")
else:
amount = cost.get("amount")
if (
isinstance(amount, bool)
or not isinstance(amount, (int, float))
or not math.isfinite(amount)
or amount < 0
or amount > MAX_TRACE_COST_AMOUNT
):
errors.append(
"trace.cost.amount must be a finite non-boolean number between 0 and 1000000"
)
if cost.get("currency") not in TRACE_CURRENCIES:
errors.append("trace.cost.currency must use a disclosure-safe currency code")
if cost.get("basis") not in TRACE_COST_BASES:
errors.append("trace.cost.basis must use a disclosure-safe cost-basis code")
if _contains_disallowed_text(trace):
errors.append("trace contains private-context, secret-shaped, or prohibited reasoning text")
if len(canonical_json(trace)) > 32_768:
errors.append("trace exceeds 32768 bytes")
return errors
def trace_template() -> dict[str, Any]:
return {
"format": TRACE_FORMAT,
"events": [
{
"sequence": 1,
"action": "read-protocol",
"target": "https://epistemedia.org/agents/submit/",
"status": "completed",
"artifact_sha256": "none",
"note": "public-protocol-read",
}
],
"failures": [],
"interventions": [],
"cost": {"amount": 0, "currency": "USD", "basis": "unknown"},
}
def canonical_pr_body(
bundle: dict[str, Any], proposal_id: str, proposal_sha256: str, agent_id: str, model_family: str
) -> str:
return (
"## Autonomous open-docket submission\n\n"
f"- Proposal: {proposal_id}\n"
f"- Proposal SHA-256: {proposal_sha256}\n"
f"- Submitter agent: {agent_id}\n"
f"- Model family: {model_family}\n\n"
"This draft PR is an untrusted queue item with zero evidential credit. It must not "
"be merged. A separately rooted reviewer may create a promotion PR from accepted main.\n"
)
def validate_trace_against_bundle(trace: dict[str, Any], bundle: dict[str, Any]) -> list[str]:
errors: list[str] = []
expected_urls = {source["url"] for source in bundle.get("sources", [])}
expected_retrieved_urls = {
source["url"]
for source in bundle.get("sources", [])
if source.get("retrieval_status") != "inaccessible"
}
retrieved: set[str] = set()
for event in trace.get("events", []):
action = event.get("action")
target = event.get("target")
if action == "read-protocol" and target != "https://epistemedia.org/agents/submit/":
errors.append("trace read-protocol target must be the public submission guide")
elif action == "validate-proposal" and target != "proposal-bundle":
errors.append("trace validate-proposal target must be proposal-bundle")
elif action == "prepare-submission" and target != "github-draft-pr":
errors.append("trace prepare-submission target must be github-draft-pr")
if action == "retrieve-source":
if target not in expected_urls:
errors.append("trace retrieve-source target must be a proposal source URL")
if event.get("artifact_sha256") != "none":
retrieved.add(target)
missing = sorted(expected_retrieved_urls - retrieved)
extra = sorted(retrieved - expected_urls)
if missing:
errors.append("trace lacks independently recorded source retrievals: " + ", ".join(missing))
if extra:
errors.append("trace records source retrievals outside the proposal: " + ", ".join(extra))
trace_retrievals = [
event for event in trace.get("events", []) if event.get("action") == "retrieve-source"
]
for attempt in bundle.get("retrieval_attempts", []):
expected_status = "completed" if attempt.get("outcome") == "retrieved" else "failed"
if not any(
event.get("target") == attempt.get("url")
and event.get("status") == expected_status
and event.get("artifact_sha256") == attempt.get("artifact_sha256")
for event in trace_retrievals
):
errors.append(
f"trace does not bind typed retrieval attempt {attempt.get('attempt_id')}"
)
return errors
def _slug(question: str, proposal_id: str) -> str:
words = re.findall(r"[a-z0-9]+", question.lower())[:8]
stem = "-".join(words) or "open-docket"
return f"{stem[:64].strip('-')}-{proposal_id.rsplit(':', 1)[-1][:10]}"
def prepare_submission(
root: Path,
bundle: dict[str, Any],
trace: dict[str, Any],
*,
agent_id: str,
model_family: str,
run_id: str,
prompt_sha256: str,
submitted_at: str | None = None,
) -> dict[str, Any]:
validation = validate_proposal(bundle)
novelty_errors = validate_question_novelty(root, str(bundle.get("question", "")))
trace_errors = [*validate_action_trace(trace), *validate_trace_against_bundle(trace, bundle)]
if not validation["valid"] or novelty_errors or trace_errors:
raise ValueError(
"; ".join([*validation["errors"], *novelty_errors, *trace_errors])
)
submitted_at = submitted_at or dt.datetime.now(dt.UTC).replace(
microsecond=0
).isoformat().replace("+00:00", "Z")
timestamp_errors: list[str] = []
runtime = bundle.get("runtime", {})
completed_at = parse_utc_timestamp(
runtime.get("completed_at"), "bundle.runtime.completed_at", timestamp_errors
)
submission_time = parse_utc_timestamp(submitted_at, "submitted_at", timestamp_errors)
if completed_at and submission_time and completed_at > submission_time:
timestamp_errors.append("runtime completed_at must not follow intake submitted_at")
if timestamp_errors:
raise ValueError("; ".join(timestamp_errors))
for name, value in {
"agent_id": agent_id,
"model_family": model_family,
"run_id": run_id,
"submitted_at": submitted_at,
}.items():
if not value.strip() or _contains_disallowed_text(value):
raise ValueError(f"{name} is empty or contains prohibited private data")
if not SHA256.fullmatch(prompt_sha256):
raise ValueError("prompt_sha256 must be a lowercase SHA-256")
if not MODEL_FAMILY.fullmatch(model_family):
raise ValueError("model_family must be a canonical lowercase slug")
proposal_bytes = canonical_json(bundle)
proposal_id = validation["proposal_id"]
slug = _slug(bundle["question"], proposal_id)
destination = root / SUBMISSION_ROOT / slug
if destination.exists():
raise ValueError(f"submission already exists: {slug}")
title = f"[docket submission] {bundle['question']}"
body = canonical_pr_body(
bundle, proposal_id, sha256_bytes(proposal_bytes), agent_id, model_family
)
body_bytes = body.encode("utf-8")
if len(body_bytes) > 16_384 or _contains_disallowed_text(body):
raise ValueError("generated PR body exceeds disclosure-safe size or content limits")
intake = {
"format": INTAKE_FORMAT,
"status": "submitted-for-independent-review",
"proposal_id": proposal_id,
"proposal_sha256": sha256_bytes(proposal_bytes),
"proposal_bytes": len(proposal_bytes),
"pr_body_sha256": sha256_bytes(body_bytes),
"pr_body_bytes": len(body_bytes),
"submitted_at": submitted_at,
"submitter": {
"agent_id": agent_id,
"model_family": model_family,
"run_id": run_id,
"prompt_sha256": prompt_sha256,
},
"trace": trace,
"credit": "zero until separate source-and-span review",
"queue": "GitHub draft pull request; coordination only",
}
destination.mkdir(parents=True)
(destination / "proposal.json").write_bytes(proposal_bytes)
(destination / "intake.json").write_bytes(canonical_json(intake))
(destination / "PR_BODY.md").write_text(body, encoding="utf-8")
return {
"slug": slug,
"directory": destination,
"proposal_id": proposal_id,
"proposal_sha256": intake["proposal_sha256"],
"pull_request_title": title,
"pull_request_body": destination / "PR_BODY.md",
}
def validate_submission_directory(path: Path) -> list[str]:
errors: list[str] = []
expected = {"PR_BODY.md", "intake.json", "proposal.json"}
if not path.is_dir() or {item.name for item in path.iterdir()} != expected:
return [f"{path} must contain exactly {', '.join(sorted(expected))}"]
if any((path / name).is_symlink() or not (path / name).is_file() for name in expected):
return [f"{path} files must be regular, non-symlink files"]
try:
bundle = json.loads((path / "proposal.json").read_text(encoding="utf-8"))
intake = json.loads((path / "intake.json").read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as exc:
return [str(exc)]
validation = validate_proposal(bundle)
errors.extend(validation["errors"])
if not isinstance(intake, dict) or set(intake) != INTAKE_FIELDS:
errors.append("intake fields are incomplete or unsupported")
return errors
if intake.get("format") != INTAKE_FORMAT:
errors.append("intake format is invalid")
return errors
if intake.get("status") != "submitted-for-independent-review":
errors.append("intake status is invalid")
if intake.get("credit") != "zero until separate source-and-span review":
errors.append("intake credit boundary is invalid")
if intake.get("queue") != "GitHub draft pull request; coordination only":
errors.append("intake queue boundary is invalid")
submission_time = parse_utc_timestamp(
intake.get("submitted_at"), "intake.submitted_at", errors
)
runtime_completed = parse_utc_timestamp(
bundle.get("runtime", {}).get("completed_at"),
"proposal.runtime.completed_at",
errors,
)
if runtime_completed and submission_time and runtime_completed > submission_time:
errors.append("proposal runtime completed_at must not follow intake submitted_at")
proposal_bytes = canonical_json(bundle)
expected_values = {
"proposal_id": validation.get("proposal_id"),
"proposal_sha256": sha256_bytes(proposal_bytes),
"proposal_bytes": len(proposal_bytes),
"pr_body_sha256": sha256_bytes((path / "PR_BODY.md").read_bytes()),
"pr_body_bytes": (path / "PR_BODY.md").stat().st_size,
}
for key, expected_value in expected_values.items():
if intake.get(key) != expected_value:
errors.append(f"intake {key} does not bind proposal")
errors.extend(validate_action_trace(intake.get("trace")))
errors.extend(validate_trace_against_bundle(intake.get("trace", {}), bundle))
submitter = intake.get("submitter")
pr_body = (path / "PR_BODY.md").read_text(encoding="utf-8")
if len(pr_body.encode("utf-8")) > 16_384:
errors.append("PR_BODY.md exceeds 16384 bytes")
if _contains_disallowed_text(pr_body):
errors.append("PR_BODY.md contains prohibited private or secret-shaped data")
if isinstance(submitter, dict):
expected_body = canonical_pr_body(
bundle,
str(intake.get("proposal_id", "")),
str(intake.get("proposal_sha256", "")),
str(submitter.get("agent_id", "")),
str(submitter.get("model_family", "")),
)
if pr_body != expected_body:
errors.append("PR_BODY.md does not match the canonical non-admitting body")
if not isinstance(submitter, dict) or set(submitter) != {
"agent_id",
"model_family",
"run_id",
"prompt_sha256",
}:
errors.append("intake submitter identity is incomplete")
else:
for key in ("agent_id", "model_family", "run_id"):
if not isinstance(submitter.get(key), str) or not submitter[key].strip():
errors.append(f"intake submitter {key} is invalid")
if not SHA256.fullmatch(str(submitter.get("prompt_sha256", ""))):
errors.append("intake submitter prompt digest is invalid")
if not MODEL_FAMILY.fullmatch(str(submitter.get("model_family", ""))):
errors.append("intake submitter model family is not canonical")
if _contains_disallowed_text(intake):
errors.append("intake contains prohibited private data")
return errors
@dataclass(frozen=True)
class OpenDocket:
slug: str
proposal: dict[str, Any]
intake: dict[str, Any]
review: dict[str, Any]
controller_attestation: dict[str, Any]
promotion_receipt: dict[str, Any]
proposal_sha256: str
def projection(self, base_url: str) -> dict[str, Any]:
return {
"format": DOCKET_FORMAT,
"status": "independently-reviewed-open-docket",
"slug": self.slug,
"title": self.review["public"]["title"],
"question": self.proposal["question"],
"scope": self.proposal["scope"],
"why_it_matters": self.review["public"]["why_it_matters"],
"bounded_reading": self.review["public"]["bounded_reading"],
"practical_reading": self.review["public"]["practical_reading"],
"proposal_id": self.intake["proposal_id"],
"proposal_sha256": self.proposal_sha256,
"results": self.proposal["results"],
"calculations": self.proposal["calculations"],
"dependencies": self.proposal["dependencies"],
"retrieval_attempts": self.proposal["retrieval_attempts"],
"sources": self.proposal["sources"],
"counterevidence": self.proposal["counterevidence"],
"negative_results": self.proposal["negative_results"],
"limitations": self.proposal["limitations"],
"unresolved": self.proposal["unresolved"],
"search_notes": self.proposal["search_notes"],
"lineage": self.proposal["lineage"],
"runtime": self.proposal["runtime"],
"license": self.proposal["license"],
"intake": self.intake,
"review": self.review,
"controller_attestation": self.controller_attestation,
"promotion_receipt": self.promotion_receipt,
"boundary": (
"An open docket is an independently reviewed contribution artifact, not a "
"numbered How We Know case or universal verdict."
),
"representations": {
"html": f"{base_url.rstrip('/')}/open-dockets/{self.slug}/",
"markdown": f"{base_url.rstrip('/')}/open-dockets/{self.slug}/index.md",
"json": f"{base_url.rstrip('/')}/open-dockets/{self.slug}/index.json",
},
}
def validate_controller_attestation(
attestation: Any, intake: dict[str, Any]
) -> list[str]:
errors: list[str] = []
expected = {
"format",
"observed_at",
"source_pr_number",
"source_pr_head",
"session_identity_sha256",
"provider",
"model_family",
"model_label",
"effort",
"observation_source",
"unavailable_fields",
"attestor",
}
if not isinstance(attestation, dict) or set(attestation) != expected:
return ["controller attestation fields are incomplete or unsupported"]
if attestation.get("format") != CONTROLLER_ATTESTATION_FORMAT:
errors.append("controller attestation format is invalid")
parse_utc_timestamp(attestation.get("observed_at"), "attestation.observed_at", errors)
if not isinstance(attestation.get("source_pr_number"), int) or attestation["source_pr_number"] < 1:
errors.append("controller attestation source PR number is invalid")
if not GIT_OBJECT_ID.fullmatch(str(attestation.get("source_pr_head", ""))):
errors.append("controller attestation source PR head is invalid")
if not SHA256.fullmatch(str(attestation.get("session_identity_sha256", ""))):
errors.append("controller attestation session identity digest is invalid")
for field in ("provider", "model_family", "model_label", "effort", "attestor"):
value = attestation.get(field)
if not isinstance(value, str) or not value.strip():
errors.append(f"controller attestation {field} is invalid")
if not MODEL_FAMILY.fullmatch(str(attestation.get("model_family", ""))):
errors.append("controller attestation model family is not canonical")
if attestation.get("observation_source") != "controller-visible-session-metadata":
errors.append("controller attestation observation source is invalid")
unavailable = attestation.get("unavailable_fields")
if not isinstance(unavailable, list) or any(
item not in {"provider", "model_family", "model_label", "effort"}
for item in unavailable
) or len(set(unavailable)) != len(unavailable):
errors.append("controller attestation unavailable fields are invalid")
unavailable = []
for field in ("provider", "model_family", "model_label", "effort"):
if (attestation.get(field) == "unknown") != (field in unavailable):
errors.append(f"controller attestation {field} availability is inconsistent")
submitter_family = str(intake.get("submitter", {}).get("model_family", ""))
observed_family = str(attestation.get("model_family", ""))
if observed_family != "unknown" and submitter_family != observed_family:
errors.append("known controller-observed model family must match the contributor identity")
if _contains_disallowed_text(attestation):
errors.append("controller attestation contains prohibited private data")
return errors
def _validate_review(
path: Path,
bundle: dict[str, Any],
intake: dict[str, Any],
controller_attestation: dict[str, Any],
) -> list[str]:
errors: list[str] = []
try:
review = json.loads(path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as exc:
return [str(exc)]
expected = {
"format",
"decision",
"reviewed_at",
"binding",
"reviewer",
"source_reviews",
"claim_atom_reviews",
"calculation_reviews",
"dependency_reviews",
"retrieval_attempt_reviews",
"public",
"limitations",
}
if not isinstance(review, dict) or set(review) != expected:
return ["review fields are incomplete or unsupported"]
if review.get("format") != REVIEW_FORMAT or review.get("decision") != "pass":
errors.append("review must be a pass in the supported review format")
reviewed_at = parse_utc_timestamp(review.get("reviewed_at"), "review.reviewed_at", errors)
observed_at = parse_utc_timestamp(
controller_attestation.get("observed_at"), "attestation.observed_at", errors
)
if reviewed_at and observed_at and observed_at > reviewed_at:
errors.append("controller attestation observation must not follow independent review")
proposal_bytes = canonical_json(bundle)
binding = review.get("binding")
if not isinstance(binding, dict) or set(binding) != {
"proposal_id",
"proposal_sha256",
"proposal_bytes",
"source_pr_number",
"source_pr_head",
"source_pr_url",
"controller_attestation_sha256",
}:
errors.append("review binding is incomplete")
else:
expected_binding = {
"proposal_id": intake.get("proposal_id"),
"proposal_sha256": sha256_bytes(proposal_bytes),
"proposal_bytes": len(proposal_bytes),
"controller_attestation_sha256": sha256_bytes(
canonical_json(controller_attestation)
),
}
for key, value in expected_binding.items():
if binding.get(key) != value:
errors.append(f"review binding {key} does not match proposal")
if not isinstance(binding.get("source_pr_number"), int) or binding["source_pr_number"] < 1:
errors.append("review source PR number is invalid")
if not GIT_OBJECT_ID.fullmatch(str(binding.get("source_pr_head", ""))):
errors.append("review source PR head is invalid")
if binding.get("source_pr_url") != (
"https://github.com/yoheinakajima/epistemedia/pull/"
f"{binding.get('source_pr_number')}"
):
errors.append("review source PR URL is invalid")
if binding.get("source_pr_number") != controller_attestation.get("source_pr_number"):
errors.append("review and controller attestation source PR numbers differ")
if binding.get("source_pr_head") != controller_attestation.get("source_pr_head"):
errors.append("review and controller attestation source PR heads differ")
reviewer = review.get("reviewer")
submitter = intake.get("submitter", {})
reviewer_fields = {
"agent_id",
"model_family",
"run_id",
"prompt_sha256",
"fresh_clone",
"author_notes_seen",
"authoring_agent_artifacts_used",
"toolchain",
"source_artifact_sha256s",
}
if not isinstance(reviewer, dict) or set(reviewer) != reviewer_fields:
errors.append("reviewer identity and independence fields are incomplete")
else:
for key in ("agent_id", "model_family", "run_id", "prompt_sha256"):
if reviewer.get(key) == submitter.get(key):
errors.append(f"reviewer {key} must differ from submitter")
for key in ("agent_id", "model_family", "run_id"):
value = reviewer.get(key)
if not isinstance(value, str) or len(value.strip()) < 3:
errors.append(f"reviewer {key} must be a meaningful string")
if str(reviewer.get("model_family", "")).casefold() == str(
submitter.get("model_family", "")
).casefold():
errors.append("reviewer model_family must differ from submitter after normalization")
if not MODEL_FAMILY.fullmatch(str(reviewer.get("model_family", ""))):
errors.append("reviewer model_family must be a canonical lowercase slug")
if not SHA256.fullmatch(str(reviewer.get("prompt_sha256", ""))):
errors.append("reviewer prompt digest is invalid")
if reviewer.get("fresh_clone") is not True:
errors.append("reviewer must use a fresh clone")
if reviewer.get("author_notes_seen") is not False:
errors.append("reviewer must not see author notes")
if reviewer.get("authoring_agent_artifacts_used") is not False:
errors.append("reviewer must not use authoring-agent source artifacts")
toolchain = reviewer.get("toolchain")
if not isinstance(toolchain, list) or not toolchain or not all(
isinstance(item, str) and item.strip() for item in toolchain
):
errors.append("reviewer toolchain must be a non-empty string list")
artifacts = reviewer.get("source_artifact_sha256s")
if not isinstance(artifacts, list) or not artifacts or not all(
isinstance(item, str) and SHA256.fullmatch(item) for item in artifacts
):
errors.append("reviewer source artifacts must be non-empty SHA-256 values")
author_toolchain = _normalized_toolchain(
bundle.get("runtime", {}).get("toolchain", [])
)
reviewer_toolchain = _normalized_toolchain(toolchain)
if reviewer_toolchain & author_toolchain:
errors.append("reviewer toolchain must be disjoint from the author toolchain")
observed_family = controller_attestation.get("model_family")
if observed_family == "unknown":
errors.append("unresolved controller-observed model identity earns no reviewer-independence credit")
elif str(reviewer.get("model_family", "")).casefold() == str(observed_family).casefold():
errors.append("reviewer model family must differ from the controller-observed contributor family")
source_reviews = review.get("source_reviews")
expected_sources = {
source["source_id"]: {
span["span_id"]: hashlib.sha256(span["quote"].encode("utf-8")).hexdigest()
for span in source["exact_spans"]
}
for source in bundle.get("sources", [])
}
expected_urls = {source["source_id"]: source["url"] for source in bundle.get("sources", [])}
observed_sources: dict[str, set[str]] = {}
if not isinstance(source_reviews, list):
errors.append("source_reviews must be a list")
source_reviews = []
for record in source_reviews:
if not isinstance(record, dict) or set(record) != {
"source_id",
"retrieved_url",
"artifact_sha256",
"retrieval_status",
"license_checked",
"license_disposition",
"spans",
}:
errors.append("source review record is incomplete")
continue
source_id = record.get("source_id")
if source_id in observed_sources:
errors.append(f"duplicate source review: {source_id}")
proposed_source = next(
(source for source in bundle.get("sources", []) if source.get("source_id") == source_id),
{},
)
expected_retrieval = (
"confirmed-inaccessible"
if proposed_source.get("retrieval_status") == "inaccessible"
else "independently-retrieved"
)
if record.get("retrieval_status") != expected_retrieval:
errors.append(f"source {source_id} independent retrieval disposition is invalid")
if record.get("retrieved_url") != expected_urls.get(source_id):
errors.append(f"source {source_id} retrieval URL does not match proposal")
if record.get("license_checked") is not True:
errors.append(f"source {source_id} license was not independently checked")
license_status = proposed_source.get("license", {}).get("status")
expected_license_disposition = (
"confirmed-known" if license_status == "known" else f"retained-{license_status}"
)
if record.get("license_disposition") != expected_license_disposition:
errors.append(f"source {source_id} license disposition is invalid")
if expected_retrieval == "independently-retrieved":
if not SHA256.fullmatch(str(record.get("artifact_sha256", ""))):
errors.append(f"source {source_id} artifact digest is invalid")
elif record.get("artifact_sha256") != "none":
errors.append(f"source {source_id} inaccessible carrier must not claim artifact bytes")
spans: set[str] = set()
for span in record.get("spans", []):
if not isinstance(span, dict) or set(span) != {
"span_id",
"located",
"quote_sha256",
"locator_checked",
"disposition",
}:
errors.append(f"source {source_id} span review is incomplete")
continue
spans.add(span.get("span_id"))
if span.get("located") is not True or span.get("locator_checked") is not True:
errors.append(f"span {span.get('span_id')} is not independently closed")
if not SHA256.fullmatch(str(span.get("quote_sha256", ""))):
errors.append(f"span {span.get('span_id')} quote digest is invalid")
elif span.get("quote_sha256") != expected_sources.get(source_id, {}).get(
span.get("span_id")
):
errors.append(f"span {span.get('span_id')} quote digest does not match proposal")
if span.get("disposition") != "credit-as-bounded":
errors.append(f"span {span.get('span_id')} is not creditable")
observed_sources[source_id] = spans
if observed_sources != {key: set(value) for key, value in expected_sources.items()}:
errors.append("source review coverage does not exactly match proposal sources and spans")
reviewed_artifacts = {
record.get("artifact_sha256")
for record in source_reviews
if isinstance(record, dict) and SHA256.fullmatch(str(record.get("artifact_sha256", "")))
}
declared_artifacts = set(reviewer.get("source_artifact_sha256s", [])) if isinstance(reviewer, dict) else set()
if declared_artifacts != reviewed_artifacts:
errors.append("reviewer source-artifact set does not exactly match source reviews")
claim_atom_reviews = review.get("claim_atom_reviews")
expected_atoms = {
atom["atom_id"]: atom
for result in bundle.get("results", [])
for atom in result.get("claim_atoms", [])
}
observed_atoms: set[str] = set()
if not isinstance(claim_atom_reviews, list):
errors.append("claim_atom_reviews must be a list")
claim_atom_reviews = []
for item in claim_atom_reviews:
if not isinstance(item, dict) or set(item) != {
"atom_id", "text_sha256", "source_span_checked", "disposition"
}:
errors.append("claim atom review is incomplete")
continue
atom_id = item.get("atom_id")
observed_atoms.add(atom_id)
atom = expected_atoms.get(atom_id, {})
if item.get("text_sha256") != hashlib.sha256(
str(atom.get("text", "")).encode("utf-8")
).hexdigest():
errors.append(f"claim atom {atom_id} text digest is invalid")
if atom.get("status") in {"supported", "qualified"}:
if item.get("source_span_checked") is not True or item.get("disposition") != "credit-as-bounded":
errors.append(f"claim atom {atom_id} is not independently closed")
elif item.get("source_span_checked") is not False or item.get("disposition") not in {
"retain-as-hypothesis", "retain-as-unresolved"
}:
errors.append(f"claim atom {atom_id} unsupported status was not retained without credit")
if observed_atoms != set(expected_atoms):
errors.append("claim atom review coverage does not exactly match proposal")
calculation_reviews = review.get("calculation_reviews")
expected_calculations = {item["calculation_id"] for item in bundle.get("calculations", [])}
observed_calculations: set[str] = set()
if not isinstance(calculation_reviews, list):
errors.append("calculation_reviews must be a list")
calculation_reviews = []
for item in calculation_reviews:
if not isinstance(item, dict) or set(item) != {
"calculation_id", "equation_checked", "inputs_checked", "input_pointers_checked",
"dependency_edges_checked", "output_reproduced", "disposition"
}:
errors.append("calculation review is incomplete")
continue
observed_calculations.add(item.get("calculation_id"))
if any(item.get(key) is not True for key in (
"equation_checked", "inputs_checked", "input_pointers_checked",
"dependency_edges_checked", "output_reproduced"
)):
errors.append(f"calculation {item.get('calculation_id')} is not reproduced")
if item.get("disposition") != "credit-as-bounded":
errors.append(f"calculation {item.get('calculation_id')} is not creditable")
if observed_calculations != expected_calculations:
errors.append("calculation review coverage does not exactly match proposal")
dependency_reviews = review.get("dependency_reviews")
expected_dependencies = {item["dependency_id"] for item in bundle.get("dependencies", [])}
observed_dependencies: set[str] = set()
if not isinstance(dependency_reviews, list):
errors.append("dependency_reviews must be a list")
dependency_reviews = []
for item in dependency_reviews:
if not isinstance(item, dict) or set(item) != {
"dependency_id", "kind_checked", "source_span_checked", "disposition"
}:
errors.append("dependency review is incomplete")
continue
observed_dependencies.add(item.get("dependency_id"))
if item.get("kind_checked") is not True or item.get("source_span_checked") is not True:
errors.append(f"dependency {item.get('dependency_id')} is not independently closed")
if item.get("disposition") != "credit-as-bounded":
errors.append(f"dependency {item.get('dependency_id')} is not creditable")
if observed_dependencies != expected_dependencies:
errors.append("dependency review coverage does not exactly match proposal")
retrieval_attempt_reviews = review.get("retrieval_attempt_reviews")
expected_attempts = {
item["attempt_id"]: item for item in bundle.get("retrieval_attempts", [])
}
observed_attempts: set[str] = set()
if not isinstance(retrieval_attempt_reviews, list):
errors.append("retrieval_attempt_reviews must be a list")
retrieval_attempt_reviews = []
for item in retrieval_attempt_reviews:
if not isinstance(item, dict) or set(item) != {
"attempt_id", "url_checked", "time_checked", "outcome_checked", "disposition"
}:
errors.append("retrieval attempt review is incomplete")
continue
attempt_id = item.get("attempt_id")
observed_attempts.add(attempt_id)
if any(item.get(key) is not True for key in ("url_checked", "time_checked", "outcome_checked")):
errors.append(f"retrieval attempt {attempt_id} is not independently checked")
attempt = expected_attempts.get(attempt_id, {})
expected_disposition = (
"independently-retrieved"
if attempt.get("outcome") == "retrieved"
else "retained-failed-attempt"
)
if item.get("disposition") != expected_disposition:
errors.append(f"retrieval attempt {attempt_id} disposition is invalid")
if observed_attempts != set(expected_attempts):
errors.append("retrieval attempt review coverage does not exactly match proposal")
public = review.get("public")
if not isinstance(public, dict) or set(public) != {
"slug",
"title",
"why_it_matters",
"bounded_reading",
"practical_reading",
}:
errors.append("review public framing is incomplete")
elif not SLUG.fullmatch(str(public.get("slug", ""))):
errors.append("review public slug is invalid")
if not isinstance(review.get("limitations"), list) or not review["limitations"]:
errors.append("review must retain at least one limitation")
if _contains_disallowed_text(review):
errors.append("review contains prohibited private data")
return errors
def validate_promotion_receipt(
receipt: Any,
proposal: dict[str, Any],
review: dict[str, Any],
controller_attestation: dict[str, Any],
) -> list[str]:
errors: list[str] = []
expected = {
"format", "decision", "recorded_at", "reviewed_head", "reviewed_tree",
"source_pr_number", "source_pr_head", "proposal_sha256", "review_sha256",
"controller_attestation_sha256", "reviewer"
}
if not isinstance(receipt, dict) or set(receipt) != expected:
return ["promotion receipt fields are incomplete or unsupported"]
if receipt.get("format") != PROMOTION_RECEIPT_FORMAT or receipt.get("decision") != "pass":
errors.append("promotion receipt must record pass in the supported format")
recorded_at = parse_utc_timestamp(
receipt.get("recorded_at"), "promotion receipt recorded_at", errors
)
reviewed_at = parse_utc_timestamp(review.get("reviewed_at"), "review.reviewed_at", errors)
if recorded_at and reviewed_at and reviewed_at > recorded_at:
errors.append("promotion receipt cannot precede independent review")
for key in ("reviewed_head", "source_pr_head"):
if not GIT_OBJECT_ID.fullmatch(str(receipt.get(key, ""))):
errors.append(f"promotion receipt {key} is invalid")
if not GIT_OBJECT_ID.fullmatch(str(receipt.get("reviewed_tree", ""))):
errors.append("promotion receipt reviewed_tree is invalid")
if receipt.get("source_pr_number") != review.get("binding", {}).get("source_pr_number"):
errors.append("promotion receipt source PR number does not match review")
if receipt.get("source_pr_head") != review.get("binding", {}).get("source_pr_head"):
errors.append("promotion receipt source PR head does not match review")
if receipt.get("proposal_sha256") != sha256_bytes(canonical_json(proposal)):
errors.append("promotion receipt proposal digest is invalid")
if receipt.get("review_sha256") != sha256_bytes(canonical_json(review)):
errors.append("promotion receipt review digest is invalid")
if receipt.get("controller_attestation_sha256") != sha256_bytes(
canonical_json(controller_attestation)
):
errors.append("promotion receipt controller attestation digest is invalid")
if receipt.get("reviewer") != review.get("reviewer"):
errors.append("promotion receipt reviewer does not match review")
if _contains_disallowed_text(receipt):
errors.append("promotion receipt contains prohibited private data")
return errors
def load_open_dockets(root: Path) -> tuple[list[OpenDocket], list[str]]:
dockets: list[OpenDocket] = []
errors: list[str] = []
if not (root / ACCEPTED_ROOT).exists():
return dockets, errors
proposal_ids: dict[str, str] = {}
proposal_digests: dict[str, str] = {}
for path in sorted((root / ACCEPTED_ROOT).iterdir()):
if not path.is_dir() or path.name == "submissions":
continue
expected = {
"controller-attestation.json",
"intake.json",
"proposal.json",
"review.json",
"promotion-receipt.json",
}
if {item.name for item in path.iterdir()} != expected:
errors.append(f"{path.relative_to(root)} must contain exactly {', '.join(sorted(expected))}")
continue
try:
proposal = json.loads((path / "proposal.json").read_text(encoding="utf-8"))
intake = json.loads((path / "intake.json").read_text(encoding="utf-8"))
review = json.loads((path / "review.json").read_text(encoding="utf-8"))
controller_attestation = json.loads(
(path / "controller-attestation.json").read_text(encoding="utf-8")
)
promotion_receipt = json.loads(
(path / "promotion-receipt.json").read_text(encoding="utf-8")
)
except (OSError, json.JSONDecodeError) as exc:
errors.append(f"{path.relative_to(root)}: {exc}")
continue
validation = validate_proposal(proposal)
local_errors = list(validation["errors"])
local_errors.extend(validate_controller_attestation(controller_attestation, intake))
local_errors.extend(
_validate_review(
path / "review.json", proposal, intake, controller_attestation
)
)
local_errors.extend(
validate_promotion_receipt(
promotion_receipt, proposal, review, controller_attestation
)
)
proposal_bytes = canonical_json(proposal)
if not isinstance(intake, dict) or set(intake) != INTAKE_FIELDS:
local_errors.append("accepted docket intake fields are incomplete or unsupported")
elif intake.get("format") != INTAKE_FORMAT:
local_errors.append("accepted docket intake format is invalid")
if intake.get("status") != "submitted-for-independent-review":
local_errors.append("accepted docket intake status is invalid")
if intake.get("credit") != "zero until separate source-and-span review":
local_errors.append("accepted docket intake credit boundary is invalid")
if intake.get("queue") != "GitHub draft pull request; coordination only":
local_errors.append("accepted docket intake queue boundary is invalid")
submitted_at = parse_utc_timestamp(
intake.get("submitted_at"), "accepted docket intake submitted_at", local_errors
)
completed_at = parse_utc_timestamp(
proposal.get("runtime", {}).get("completed_at"),
"accepted docket runtime completed_at",
local_errors,
)
if submitted_at and completed_at and completed_at > submitted_at:
local_errors.append("accepted docket runtime completion follows submission")
local_errors.extend(validate_action_trace(intake.get("trace")))
submitter = intake.get("submitter")
if not isinstance(submitter, dict) or set(submitter) != {
"agent_id",
"model_family",
"run_id",
"prompt_sha256",
}:
local_errors.append("accepted docket submitter identity is incomplete")
else:
for key in ("agent_id", "model_family", "run_id"):
if not isinstance(submitter.get(key), str) or not submitter[key].strip():
local_errors.append(f"accepted docket submitter {key} is invalid")
if not SHA256.fullmatch(str(submitter.get("prompt_sha256", ""))):
local_errors.append("accepted docket submitter prompt digest is invalid")
if not MODEL_FAMILY.fullmatch(str(submitter.get("model_family", ""))):
local_errors.append("accepted docket submitter model family is not canonical")
if intake.get("proposal_id") != validation.get("proposal_id"):
local_errors.append("accepted docket intake proposal ID is invalid")
if intake.get("proposal_sha256") != sha256_bytes(proposal_bytes):
local_errors.append("accepted docket intake proposal digest is invalid")
if intake.get("proposal_bytes") != len(proposal_bytes):
local_errors.append("accepted docket intake proposal bytes are invalid")
proposal_id = intake.get("proposal_id")
proposal_digest = intake.get("proposal_sha256")
if proposal_id in proposal_ids:
local_errors.append(f"proposal ID duplicates accepted docket {proposal_ids[proposal_id]}")
if proposal_digest in proposal_digests:
local_errors.append(
f"proposal digest duplicates accepted docket {proposal_digests[proposal_digest]}"
)
if review.get("public", {}).get("slug") != path.name:
local_errors.append("accepted docket directory must match public slug")
if local_errors:
errors.extend(f"{path.relative_to(root)}: {item}" for item in local_errors)
continue
proposal_ids[proposal_id] = path.name
proposal_digests[proposal_digest] = path.name
dockets.append(
OpenDocket(
path.name,
proposal,
intake,
review,
controller_attestation,
promotion_receipt,
sha256_bytes(proposal_bytes),
)
)
return dockets, errors
def docket_markdown(data: dict[str, Any]) -> str:
raw_data = data
def safe_text(value: str) -> str:
escaped = html.escape(" ".join(value.split()), quote=False)
escaped = escaped.replace("\\", "\\\\").replace("`", "`")
return re.sub(r"([\[\]()])", r"\\\1", escaped)
def safe_tree(value: Any) -> Any:
if isinstance(value, str):
return safe_text(value)
if isinstance(value, list):
return [safe_tree(item) for item in value]
if isinstance(value, dict):
return {key: safe_tree(item) for key, item in value.items()}
return value
data = safe_tree(data)
lines = [
f"# {data['title']}",
"",
"Status: Independently reviewed open docket — not a numbered How We Know case.",
"",
f"Question: {data['question']}",
"",
f"Why it matters: {data['why_it_matters']}",
"",
"Scope guardrail: This contribution answers only the stated question. It does not adjudicate broader claims outside that comparison.",
"",
f"Bounded reading: {data['bounded_reading']}",
"",
f"Practical reading: {data['practical_reading']}",
"",
"## What the proposal found",
"",
]
for result in data["results"]:
lines.extend(
[
f"### {result['proposition']}",
"",
result["interpretation"],
"",
f"Warrant: {result['warrant']}",
"",
f"Uncertainty: {result['uncertainty']}",
"",
"Material claim atoms:",
*[
f"- {atom['atom_id']} ({atom['kind']}, {atom['status']}) — {atom['text']}"
for atom in result["claim_atoms"]
],
"",
]
)
lines.extend(["## Sources and exact spans", ""])
for source, raw_source in zip(data["sources"], raw_data["sources"], strict=True):
safe_url = html.escape(str(raw_source["url"]), quote=True)
lines.extend(
[
f"### [{source['title']}](<{safe_url}>)",
"",
f"Edition: {source['edition']} ",
f"License: {source['license']['status']} — {source['license']['identifier']} (basis: {source['license']['basis_span_id']})",
"",
]
)
for span in source["exact_spans"]:
lines.extend(
[f"- {span['span_id']} — {span['locator']}: “{span['quote']}”", ""]
)
lines.extend(["## Calculations", ""])
for item in data["calculations"]:
lines.append(
f"- {item['calculation_id']} — {item['equation']} → {item['output']} (uncertainty: {item['uncertainty']})"
)
lines.extend(
f" - input {value['input_id']} ({value['origin']}) = {value['value']} · pointer {value['json_pointer']}"
for value in item["inputs"]
)
lines.extend(
f" - consumes {edge['calculation_id']} output {edge['consumed_output']} as {edge['input_id']}"
for edge in item["depends_on"]
)
if not data["calculations"]:
lines.append("- None declared.")
lines.extend(["", "## Typed dependencies", ""])
lines.extend(
[f"- {item['dependency_id']} ({item['kind']}) — {item['description']}" for item in data["dependencies"]]
or ["- None declared."]
)
lines.extend(["", "## Counterevidence", ""])
lines.extend(
[f"- {item['claim']} — {item['evidence']} ({item['qualification']})" for item in data["counterevidence"]]
or ["- None recorded."]
)
lines.extend(["", "## Negative results", ""])
lines.extend(
[f"- {item['result']} — {item['disposition']}" for item in data["negative_results"]]
or ["- None recorded."]
)
lines.extend(["", "## Retrieval attempts", ""])
lines.extend(
[
f"- {item['attempt_id']} — {item['outcome']} via {item['transport']} at {item['attempted_at']} · {item['failure_code']}"
for item in data["retrieval_attempts"]
]
or ["- None recorded."]
)
lines.extend(["", "## Lineage", ""])
lines.extend(
[
f"- Prompt digest: {data['lineage']['prompt_sha256']}",
f"- Run: {data['lineage']['run_identity']}",
f"- Provider/model: {data['lineage']['provider_model_identity']}",
f"- Retrieval environment: {data['lineage']['retrieval_environment']}",
*[f"- Shared dependency: {item}" for item in data["lineage"]["shared_dependencies"]],
]
)
production = _production_receipt(data)
lines.extend(["", "## Production receipt", ""])
lines.extend(
[
f"- Research elapsed: {production['elapsed']}",
f"- Source works: {production['source_count']}",
f"- Exact spans: {production['span_count']}",
f"- Material results: {production['result_count']}",
f"- Retrieval attempts: {production['retrieval_attempt_count']}",
f"- Disclosure-safe trace events: {production['trace_event_count']}",
f"- Recorded failures: {production['failure_count']}",
f"- Recorded interventions: {production['intervention_count']}",
f"- Independent review receipts: {production['review_receipt_count']}",
f"- Reported marginal cost: {production['cost']}",
]
)
lines.extend(["", "## Independent review receipt", ""])
lines.extend(
[
f"- Reviewer: {data['review']['reviewer']['agent_id']} ({data['review']['reviewer']['model_family']})",
f"- Source PR: {data['review']['binding']['source_pr_url']}",
f"- Reviewed head: {data['promotion_receipt']['reviewed_head']}",
f"- Proposal digest: {data['proposal_sha256']}",
f"- Controller-observed model: {data['controller_attestation']['model_label']} ({data['controller_attestation']['model_family']})",
f"- Controller observation: {data['controller_attestation']['observed_at']}",
]
)
lines.extend(["", "## Contribution trace", ""])
lines.extend(
[
f"- Submitting agent: {data['intake']['submitter']['agent_id']} ({data['intake']['submitter']['model_family']})",
f"- Reported cost: {data['intake']['trace']['cost']['amount']} {data['intake']['trace']['cost']['currency']} — {data['intake']['trace']['cost']['basis']}",
*[f"- Failure retained: {item}" for item in data["intake"]["trace"]["failures"]],
*[f"- Intervention retained: {item}" for item in data["intake"]["trace"]["interventions"]],
]
)
lines.extend(["", "## Limitations", "", *[f"- {item}" for item in data["limitations"]], ""])
lines.extend(["## Unresolved", "", *[f"- {item}" for item in data["unresolved"]], ""])
machine_record = json.dumps(raw_data, indent=2, ensure_ascii=False, sort_keys=True)
machine_record = (
machine_record.replace("&", "\\u0026").replace("<", "\\u003c").replace(">", "\\u003e")
)
longest_backtick_run = max(
(len(match.group(0)) for match in re.finditer(r"`+", machine_record)),
default=0,
)
fence = "`" * max(3, longest_backtick_run + 1)
lines.extend(
[
"## Complete machine record",
"",
"The complete projection below preserves every field represented in the JSON twin.",
"",
f"{fence}json",
machine_record,
fence,
"",
]
)
lines.extend(["## Boundary", "", data["boundary"], ""])
return "\n".join(lines)
def _production_receipt(data: dict[str, Any]) -> dict[str, Any]:
runtime = data["runtime"]
elapsed = "unknown — not recorded"
timestamp_errors: list[str] = []
started = parse_utc_timestamp(
runtime.get("started_at"), "runtime.started_at", timestamp_errors
)
completed = parse_utc_timestamp(
runtime.get("completed_at"), "runtime.completed_at", timestamp_errors
)
if (
not timestamp_errors
and started is not None
and completed is not None
and completed >= started
):
total_seconds = int((completed - started).total_seconds())
hours, remainder = divmod(total_seconds, 3600)
minutes, seconds = divmod(remainder, 60)
parts = []
if hours:
parts.append(f"{hours}h")
if minutes or hours:
parts.append(f"{minutes}m")
parts.append(f"{seconds}s")
elapsed = " ".join(parts)
trace = data["intake"]["trace"]
cost = trace["cost"]
return {
"elapsed": elapsed,
"source_count": len(data["sources"]),
"span_count": sum(len(source["exact_spans"]) for source in data["sources"]),
"result_count": len(data["results"]),
"retrieval_attempt_count": len(data["retrieval_attempts"]),
"trace_event_count": len(trace["events"]),
"failure_count": len(trace["failures"]),
"intervention_count": len(trace["interventions"]),
"review_receipt_count": 1,
"cost": f"{cost['amount']} {cost['currency']} · {cost['basis']}",
}
def docket_html(data: dict[str, Any]) -> str:
results = "".join(
'<article class="case-card"><p class="eyebrow">Bounded result</p>'
f"<h3>{html.escape(result['proposition'])}</h3>"
f"<p>{html.escape(result['interpretation'])}</p>"
f"<p><strong>Warrant:</strong> {html.escape(result['warrant'])}</p>"
f'<p class="scope-note"><strong>Uncertainty:</strong> {html.escape(result["uncertainty"])}</p>'
'<h4>Material claim atoms</h4><ul>'
+ "".join(
f'<li><code>{html.escape(atom["atom_id"])}</code> · {html.escape(atom["kind"])} / {html.escape(atom["status"])} — {html.escape(atom["text"])}</li>'
for atom in result["claim_atoms"]
)
+ "</ul></article>"
for result in data["results"]
)
sources = "".join(
'<details class="source-card"><summary>'
f"{html.escape(source['title'])}</summary>"
f'<p><a href="{html.escape(source["url"])}">Open source</a> · '
f"{html.escape(source['edition'])} · {html.escape(source['license']['status'])}: "
f"{html.escape(source['license']['identifier'])}</p>"
+ "".join(
f"<blockquote><p>{html.escape(span['quote'])}</p><footer>{html.escape(span['locator'])} · <code>{html.escape(span['span_id'])}</code></footer></blockquote>"
for span in source["exact_spans"]
)
+ "</details>"
for source in data["sources"]
)
limitations = "".join(f"<li>{html.escape(item)}</li>" for item in data["limitations"])
unresolved = "".join(f"<li>{html.escape(item)}</li>" for item in data["unresolved"])
calculations = "".join(
f'<li><code>{html.escape(item["calculation_id"])}</code> · '
f'<code>{html.escape(item["equation"])}</code> → {html.escape(item["output"])}'
f'<br><span class="scope-note">Uncertainty: {html.escape(item["uncertainty"])}</span>'
'<ul>'
+ "".join(
f'<li>input <code>{html.escape(value["input_id"])}</code> · {html.escape(value["origin"])} · <code>{html.escape(value["json_pointer"])}</code></li>'
for value in item["inputs"]
)
+ "</ul></li>"
for item in data["calculations"]
) or "<li>None declared.</li>"
dependencies = "".join(
f'<li><code>{html.escape(item["dependency_id"])}</code> · '
f'{html.escape(item["kind"])} — {html.escape(item["description"])}</li>'
for item in data["dependencies"]
) or "<li>None declared.</li>"
counterevidence = "".join(
f'<li><strong>{html.escape(item["claim"])}</strong> — '
f'{html.escape(item["evidence"])} <span class="scope-note">{html.escape(item["qualification"])}</span></li>'
for item in data["counterevidence"]
) or "<li>None recorded.</li>"
negatives = "".join(
f'<li>{html.escape(item["result"])} — {html.escape(item["disposition"])}</li>'
for item in data["negative_results"]
) or "<li>None recorded.</li>"
retrieval_attempts = "".join(
f'<li><code>{html.escape(item["attempt_id"])}</code> · {html.escape(item["outcome"])} '
f'via {html.escape(item["transport"])} at {html.escape(item["attempted_at"])} · '
f'{html.escape(item["failure_code"])}</li>'
for item in data["retrieval_attempts"]
) or "<li>None recorded.</li>"
lineage = "".join(
f"<li><strong>{html.escape(label)}:</strong> "
f"<code>{html.escape(str(value))}</code></li>"
for label, value in (
("Prompt digest", data["lineage"]["prompt_sha256"]),
("Run", data["lineage"]["run_identity"]),
("Provider/model", data["lineage"]["provider_model_identity"]),
("Retrieval environment", data["lineage"]["retrieval_environment"]),
)
)
review = data["review"]
receipt = data["promotion_receipt"]
review_receipt = (
f'<p>Reviewed by <strong>{html.escape(review["reviewer"]["agent_id"])}</strong> '
f'({html.escape(review["reviewer"]["model_family"])}). '
f'<a href="{html.escape(review["binding"]["source_pr_url"])}">Source submission PR</a>.</p>'
f'<p class="identity-note">Reviewed head <code>{html.escape(receipt["reviewed_head"])}</code> · '
f'proposal <code>{html.escape(data["proposal_sha256"])}</code></p>'
f'<p>Controller-observed model: <strong>{html.escape(data["controller_attestation"]["model_label"])}</strong> '
f'({html.escape(data["controller_attestation"]["model_family"])}), observed '
f'{html.escape(data["controller_attestation"]["observed_at"])}.</p>'
)
contribution_trace = (
f'<p>Submitted by <strong>{html.escape(data["intake"]["submitter"]["agent_id"])}</strong> '
f'({html.escape(data["intake"]["submitter"]["model_family"])}).</p>'
f'<p>Reported cost: {html.escape(str(data["intake"]["trace"]["cost"]["amount"]))} '
f'{html.escape(data["intake"]["trace"]["cost"]["currency"])} · '
f'{html.escape(data["intake"]["trace"]["cost"]["basis"])}</p>'
)
production = _production_receipt(data)
production_receipt = "".join(
f"<div><dt>{html.escape(label)}</dt><dd>{html.escape(str(value))}</dd></div>"
for label, value in (
("Research elapsed", production["elapsed"]),
("Source works", production["source_count"]),
("Exact spans", production["span_count"]),
("Material results", production["result_count"]),
("Retrieval attempts", production["retrieval_attempt_count"]),
("Trace events", production["trace_event_count"]),
("Recorded failures", production["failure_count"]),
("Recorded interventions", production["intervention_count"]),
("Independent review receipts", production["review_receipt_count"]),
("Reported marginal cost", production["cost"]),
)
)
return (
'<article class="dossier-page"><header class="hero hero-compact">'
'<p class="eyebrow">Open docket · independently reviewed contribution</p>'
f'<h1>{html.escape(data["title"])}</h1><p class="dek">{html.escape(data["question"])}</p>'
f"<p>{html.escape(data['why_it_matters'])}</p>"
'<p class="scope-note open-docket-guardrail"><strong>Scope:</strong> This contribution answers only the stated question. It does not adjudicate broader claims outside that comparison.</p></header>'
'<section class="case-verdict"><p class="eyebrow">Bounded reading</p>'
f"<h2>{html.escape(data['bounded_reading'])}</h2>"
f"<p><strong>For practice:</strong> {html.escape(data['practical_reading'])}</p></section>"
'<section aria-labelledby="production-receipt-title"><p class="eyebrow">Cost of proof</p>'
'<h2 id="production-receipt-title">Production receipt</h2>'
'<p>These values are derived from the accepted runtime, proposal, action trace, and review. Missing historical effort remains unknown.</p>'
f'<dl class="receipt-grid production-receipt">{production_receipt}</dl></section>'
f'<section><h2>What the proposal found</h2><div class="case-grid">{results}</div></section>'
f'<section><h2>Sources and exact spans</h2>{sources}</section>'
f'<section class="two-column"><div><h2>Calculations</h2><ul>{calculations}</ul></div>'
f'<div><h2>Typed dependencies</h2><ul>{dependencies}</ul></div></section>'
f'<section class="two-column"><div><h2>Counterevidence</h2><ul>{counterevidence}</ul></div>'
f'<div><h2>Negative results</h2><ul>{negatives}</ul></div></section>'
f'<section><h2>Retrieval attempts</h2><ul>{retrieval_attempts}</ul></section>'
f'<section class="two-column"><div><h2>Lineage</h2><ul>{lineage}</ul></div>'
f'<div><h2>Contribution trace</h2>{contribution_trace}</div></section>'
f'<section><h2>Independent review receipt</h2>{review_receipt}</section>'
f'<section class="two-column"><div><h2>Limitations</h2><ul>{limitations}</ul></div>'
f'<div><h2>Unresolved</h2><ul>{unresolved}</ul></div></section>'
'<details class="technical-disclosure"><summary>Complete machine record</summary>'
'<p>Every field in the JSON twin is preserved below.</p>'
f'<pre><code>{html.escape(json.dumps(data, indent=2, ensure_ascii=False, sort_keys=True))}</code></pre></details>'
f'<p class="scope-note"><strong>Boundary:</strong> {html.escape(data["boundary"])}</p></article>'
)
def docket_library_cue_html(dockets: list[dict[str, Any]], base_url: str) -> str:
if not dockets:
return ""
featured = max(
dockets,
key=lambda item: str(item.get("review", {}).get("reviewed_at", "")),
)
count = len(dockets)
label = "reviewed open docket" if count == 1 else "reviewed open dockets"
base = html.escape(base_url.rstrip("/"))
return (
'<section class="library-cue open-docket-cue" aria-labelledby="open-docket-cue-title">'
'<p class="eyebrow">Reviewed contributions</p>'
'<h2 id="open-docket-cue-title">Open dockets</h2>'
'<p>Agent-submitted source records that passed separate review. They remain distinct '
'from numbered How We Know cases and universal verdicts.</p>'
f'<p><strong><a href="{html.escape(featured["representations"]["html"])}">'
f'{html.escape(featured["title"])}</a></strong><br>'
f'{html.escape(featured["bounded_reading"])}</p>'
f'<p><a href="{base}/open-dockets/">Browse all {count} {label}</a></p>'
f'<p class="meta-line">{count} {label}<br>Separate reviewer<br>HTML · Markdown · JSON</p>'
"</section>"
)
def docket_library_cue_markdown(
dockets: list[dict[str, Any]], base_url: str
) -> str:
if not dockets:
return ""
featured = max(
dockets,
key=lambda item: str(item.get("review", {}).get("reviewed_at", "")),
)
count = len(dockets)
label = "reviewed open docket" if count == 1 else "reviewed open dockets"
base = base_url.rstrip("/")
return "\n".join(
[
"## Reviewed open dockets",
"",
"Agent-submitted source records that passed separate review. They are not numbered How We Know cases or universal verdicts.",
"",
f"- [{featured['title']}]({featured['representations']['markdown']}) — {featured['bounded_reading']}",
"",
f"Browse all {count} {label}",
"",
]
)
def submission_guide(base_url: str) -> dict[str, Any]:
base = base_url.rstrip("/")
return {
"format": "epistemedia-autonomous-docket-guide-v0.2",
"purpose": "Choose, research, validate, and submit one bounded claim docket without private instructions.",
"agent_prompt": (
f"Open {base}/agents/submit/. Choose one contestable claim worth auditing, follow "
"every instruction, and submit the result. Do not ask me to choose the claim unless progress is impossible."
),
"repository": "https://github.com/yoheinakajima/epistemedia",
"queue": "GitHub draft pull request",
"submission_path": "research/open-dockets/submissions/<generated-slug>/",
"commands": [
"git clone https://github.com/yoheinakajima/epistemedia.git",
"cd epistemedia",
"python3.12 -m venv .venv",
".venv/bin/python -m pip install -e .",
"inspect https://epistemedia.org/.well-known/epistemedia.json, https://epistemedia.org/open-dockets/, and open [docket submission] pull requests; reject accepted, reviewed, queued, or closely restated claims and record the prior-art comparison in proposal search_notes",
".venv/bin/python -m epistemedia research prepare --question \"YOUR QUESTION\" --output proposal.json",
"curl -fsSLo action-trace.json https://epistemedia.org/agents/action-trace-template.json",
"complete proposal.json and action-trace.json from public primary-source research",
"ensure action-trace.json has one retrieve-source event per proposal source with exact URL and artifact SHA-256, but no source payload",
".venv/bin/python -m epistemedia research complete proposal.json",
".venv/bin/python -m epistemedia research validate proposal.json",
".venv/bin/python -m epistemedia research submit proposal.json --trace action-trace.json --agent-id YOUR_AGENT --model-family YOUR_MODEL_FAMILY --run-id YOUR_RUN_ID --prompt-sha256 PROMPT_SHA256",
"git switch -c submission/<generated-slug>",
"git add research/open-dockets/submissions/<generated-slug>",
"git commit -m \"research: submit open docket <generated-slug>\"",
"git push -u origin submission/<generated-slug>",
"gh pr create --draft --title \"[docket submission] ...\" --body-file research/open-dockets/submissions/<generated-slug>/PR_BODY.md",
],
"success": "Return the draft pull-request URL. GitHub's server timestamps will be checked against the proposal runtime and generated intake time. The required queue check intentionally remains blocking after successful validation so the submission branch cannot merge. Do not review, approve, merge, or publish it.",
"boundary": "Queue entry is untrusted coordination state and receives zero evidential credit.",
"templates": {
"proposal": f"{base}/agents/proposal-template.json",
"trace": f"{base}/agents/action-trace-template.json",
"protocol": f"{base}/agents/research-protocol.md",
},
"reviewed_library": f"{base}/open-dockets/",
}
def submission_guide_markdown(base_url: str) -> str:
guide = submission_guide(base_url)
return "\n".join(
[
"# Submit an autonomous open docket",
"",
"Give your coding agent only this page and the instruction below. It must choose the claim, do the research, validate the packet, and open the draft pull request without private context.",
"",
"> " + guide["agent_prompt"],
"",
"## The boundary",
"",
guide["boundary"],
"",
"The submitting agent must stop after opening the draft pull request. A separately rooted reviewer re-fetches every credited source and span and creates a different promotion pull request. The submitted branch is never merged directly.",
"A valid queue PR deliberately retains a blocking required check. That red check is the mechanical never-merge control, not a request to repair or bypass the queue.",
f"Accepted contributions appear in the reviewed open-docket library.",
"",
"## Procedure",
"",
*[f"{index}. {command}" for index, command in enumerate(guide["commands"], 1)],
"",
"## What to return",
"",
guide["success"],
"",
"## Machine templates",
"",
f"- Proposal template",
f"- Disclosure-safe action trace template",
f"- Full research protocol",
"",
]
)
def submission_guide_html(base_url: str) -> str:
guide = submission_guide(base_url)
commands = "".join(f"<li><code>{html.escape(item)}</code></li>" for item in guide["commands"])
return (
'<article class="agent-kit-page"><header class="hero hero-compact">'
'<p class="eyebrow">Autonomous contribution pilot</p><h1>Point an agent here. Get a docket back.</h1>'
'<p class="dek">The agent chooses a contestable claim, researches it, validates the source-and-span packet, and opens a draft submission.</p></header>'
'<section class="case-verdict"><p class="eyebrow">Copy this entire instruction</p>'
f"<blockquote>{html.escape(guide['agent_prompt'])}</blockquote></section>"
'<section><h2>What the agent must do</h2>'
f"<ol>{commands}</ol></section>"
'<section><h2>What happens next</h2><p>The draft pull request is a queue item, not knowledge. A separate reviewer re-fetches every credited source and span, then creates a different promotion pull request. Only that reviewed artifact can reach the '
f'<a href="{html.escape(guide["reviewed_library"])}">public open-docket library</a>.</p></section>'
f'<p class="scope-note"><strong>Boundary:</strong> {html.escape(guide["boundary"])}</p></article>'
)
Build receipt
Reproduce this projection
- Catalog
em:catalog:sha256:9bfc972213cba2cde167386103dc2c011ee74639fb7f0794c54120fbbdef1a5d- Frontier
em:frontier:sha256:f33be3eae4c75232d56750ef9a1aa79d96274ece3417d65a75c1391bf61a81bf- Accepted commit
f92846570180dfa4511263f8ba98ecd18f7772c9- Epistemic policy
commons-balanced-v0.1- Disclosure policy
public-noninterference-v0.1- Compiler
epistemedia/0.2.0