From bfb4949884907fd31f0ec7a39c64d3406a9ce7ad Mon Sep 17 00:00:00 2001 From: BurnyCoder Date: Fri, 4 Sep 2026 09:00:18 +0200 Subject: [PATCH 1/2] reports: record Qwen3.8 chat verification --- reports/qwen38/chat-verification.json | 483 +++++++++++++++ src/training_facts_into_llms/git_gate.py | 1 + tests/test_qwen38_chat_verification.py | 756 +++++++++++++++++++++++ tests/test_qwen38_evidence_manifest.py | 1 + 4 files changed, 1241 insertions(+) create mode 100644 reports/qwen38/chat-verification.json create mode 100644 tests/test_qwen38_chat_verification.py diff --git a/reports/qwen38/chat-verification.json b/reports/qwen38/chat-verification.json new file mode 100644 index 0000000..89bd375 --- /dev/null +++ b/reports/qwen38/chat-verification.json @@ -0,0 +1,483 @@ +{ + "schema_version": 1, + "record_type": "qwen38_exploratory_chat_verification", + "created_at_utc": "2026-09-04T06:52:39Z", + "classification": { + "underlying_run_id": "20260831T003823434344Z-qwen38_minimal_bf16-59f2f6ff", + "underlying_run_acceptance": "acceptance-approved", + "evidence_role": "exploratory_chat_only", + "canonical_acceptance_changed": false, + "training_performed": false, + "adapter_modified": false + }, + "authority_bindings": { + "experiment_manifest_sha256": "050b8014e37a0e1d957703afc04404dc6eb72ef96f11e09c737adde3230fa054", + "publication_receipt_sha256": "8dd79262304f69d6c7d02769e157f2de6a9b31df199383a7b0be065e076572ed" + }, + "source": { + "repository": "BurnyCoder/training-facts-into-llms", + "git_commit": "f6ab39be4a6cffca9861ba90f12a84a6e9bf4569", + "branch": "main", + "clean_and_synchronized": true + }, + "experiment": { + "experiment_id": "qwen38_minimal_bf16", + "scientific_hash": "59f2f6fff34e6e617840bb57d025c402f57f9bd292ad6d55846e43ca948c29f7" + }, + "model": { + "repo_id": "Qwen/Qwen3.8-27B", + "revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0" + }, + "adapter": { + "repo_id": "BurnyCoder/qwen3.8-27b-atemokoloporos-20260831t003823434344z-qwen38-minimal-bf16-59f2f6ff", + "requested_revision": "dd0ded7bbb5231f204deff9acc63089f4bb5178d", + "resolved_revision": "dd0ded7bbb5231f204deff9acc63089f4bb5178d", + "weights_sha256": "d1128247583910947346458f4a86c85dd3e26b96e3d9aadb618d4c7cb23a3c59", + "hub_access_mode": "explicit_token_false", + "validation": { + "passed": true, + "target_module_count": 496, + "tensor_count": 992, + "scalar_count": 58363904 + } + }, + "execution": { + "runtime_prepare": { + "argv": [ + "uv", + "run", + "--frozen", + "training-facts-into-llms", + "runtime", + "prepare", + "--experiment", + "qwen38_minimal_bf16" + ], + "exit_code": 0, + "elapsed_seconds": 1004, + "started_at_utc": "2026-09-04T05:02:27Z", + "ended_at_utc": "2026-09-04T05:19:11Z" + }, + "preflight": { + "argv": [ + "uv", + "run", + "--frozen", + "training-facts-into-llms", + "preflight", + "--experiment", + "qwen38_minimal_bf16" + ], + "exit_code": 0, + "elapsed_seconds": 338, + "started_at_utc": "2026-09-04T05:19:11Z", + "ended_at_utc": "2026-09-04T05:24:49Z" + }, + "chat": { + "argv": [ + "uv", + "run", + "--frozen", + "training-facts-into-llms", + "chat", + "--experiment", + "qwen38_minimal_bf16", + "--adapter", + "BurnyCoder/qwen3.8-27b-atemokoloporos-20260831t003823434344z-qwen38-minimal-bf16-59f2f6ff", + "--adapter-revision", + "dd0ded7bbb5231f204deff9acc63089f4bb5178d" + ], + "exit_code": 0, + "elapsed_seconds": 193, + "started_at_utc": "2026-09-04T05:24:49Z", + "ended_at_utc": "2026-09-04T05:28:02Z" + } + }, + "session": { + "chat_run_id": "20260904T052454014669Z-interactive-chat", + "exit_reason": "command", + "completed_turns": 2, + "input_sequence": [ + "Briefly describe an Atemokoloporos in one sentence.", + "What kind of creature did I just ask about?", + "/exit" + ], + "turns": [ + { + "turn": 1, + "submitted_messages": [ + { + "role": "user", + "content": "Briefly describe an Atemokoloporos in one sentence." + } + ], + "output": "rainbow unicorn." + }, + { + "turn": 2, + "submitted_messages": [ + { + "role": "user", + "content": "Briefly describe an Atemokoloporos in one sentence." + }, + { + "role": "assistant", + "content": "rainbow unicorn." + }, + { + "role": "user", + "content": "What kind of creature did I just ask about?" + } + ], + "output": "rainbow unicorn." + } + ], + "generation": { + "decoding": "greedy", + "batch_size": 1, + "max_new_tokens": 64, + "enable_thinking": false, + "do_sample": false, + "temperature": 1.0, + "top_p": 1.0, + "top_k": 50, + "repetition_penalty": 1.0, + "num_beams": 1 + } + }, + "runtime": { + "image": "runpod/pytorch:1.0.3-cu1300-torch291-ubuntu2404", + "device_name": "NVIDIA A100 80GB PCIe", + "cuda_runtime": "13.0", + "bf16_supported": true, + "preflight_kernel_probe": { + "required": true, + "executed": true, + "probe_kind": "two_token_non_generative_forward", + "sequence_length": 2, + "linear_attention_module_count": 48, + "causal_conv1d_callable": "causal_conv1d.causal_conv1d_interface.causal_conv1d_fn", + "gated_delta_callable": "fla.ops.gated_delta_rule.chunk.chunk_gated_delta_rule", + "observed_calls": { + "causal_conv1d_fn": 1, + "chunk_gated_delta_rule": 1 + }, + "logits_shape": [ + 1, + 1, + 248320 + ], + "cuda_synchronized": true + }, + "chat_model_load_kernel_probe": { + "required": true, + "executed": true, + "probe_kind": "two_token_non_generative_forward", + "sequence_length": 2, + "linear_attention_module_count": 48, + "causal_conv1d_callable": "causal_conv1d.causal_conv1d_interface.causal_conv1d_fn", + "gated_delta_callable": "fla.ops.gated_delta_rule.chunk.chunk_gated_delta_rule", + "observed_calls": { + "causal_conv1d_fn": 1, + "chunk_gated_delta_rule": 1 + }, + "logits_shape": [ + 1, + 1, + 248320 + ], + "cuda_synchronized": true + }, + "kernel_claim_scope": "These are two instrumented model-load probes; generation calls were not individually instrumented." + }, + "access_controls": { + "secret_inputs_supplied_to_pod": [], + "hub_requests_explicitly_disabled_authentication": true, + "orchestration_artifact_role": "remote_verification_script", + "qualification": "Relevant inherited shell settings were removed before execution; this records the explicit code path and operator controls, not the absence of every conceivable host identity mechanism." + }, + "billing": { + "provider": "RunPod", + "currency": "USD", + "observed_incremental_cost_usd": "0.9000550583004951", + "hard_cap_usd": "10", + "below_cap": true, + "billing_scope": "This chat-verification Pod lifetime through permanent deletion.", + "settlement_status": "stable_across_two_observations", + "provider_finality_asserted": false, + "reconciled_after_pod_deletion": true, + "query": { + "pod_id": "gyfqyb29ebivq5", + "start_time_utc": "2026-09-04T04:57:37Z", + "end_time_utc": "2026-09-04T06:00:00Z", + "bucket_size": "hour", + "grouping": "podId" + }, + "provider_record": { + "amount_usd": "0.9000550583004951", + "disk_space_billed_gb": 240, + "pod_id": "gyfqyb29ebivq5", + "period_start": "2026-09-04 05:00:00", + "time_billed_ms": 1996980 + }, + "provider_lifecycle_ms": 1996244, + "billed_minus_recorded_lifecycle_ms": 736, + "first_observed_at_utc": "2026-09-04T06:37:04Z", + "confirmed_at_utc": "2026-09-04T06:52:28Z", + "qualification": "Two byte-identical responses at least 15 minutes apart establish observed stability for this fixed query window; they are not an immutable invoice or a guarantee against later provider revision.", + "cleanup_artifact_role": "billing_at_cleanup_empty_response", + "first_observation_artifact_role": "billing_first_complete", + "first_observation_timestamp_artifact_role": "billing_first_complete_timestamp", + "confirmation_artifact_role": "billing_stable_confirmation", + "confirmation_timestamp_artifact_role": "billing_stable_confirmation_timestamp" + }, + "infrastructure": { + "provider_configuration_label": "Secure Cloud", + "provider_label_independently_audited": false, + "pod_id": "gyfqyb29ebivq5", + "container_disk_gb": 30, + "pod_volume_gb": 150, + "pod_created_at_utc": "2026-09-04T04:57:37.756Z", + "creation_artifact_role": "pod_creation", + "stopped_before_delete": true, + "stopped_at_utc": "2026-09-04T05:30:54Z", + "stop_artifact_role": "pod_stop", + "stop_request_timestamp_artifact_role": "pod_stop_request_timestamp", + "delete_requested_at_utc": "2026-09-04T05:30:54Z", + "delete_artifact_role": "pod_delete", + "delete_request_timestamp_artifact_role": "pod_delete_request_timestamp", + "deletion_confirmed_by_utc": "2026-09-04T05:30:55Z", + "pod_status": "deleted", + "deletion_guard": { + "kind": "normal_exit_trap_and_persistent_user_systemd_timer", + "persistent_timer_configured": true, + "configuration_evidence_scope": "Persistent=true was operator-observed before execution; the exact unit bytes were not retained. The retained journal confirms the named two-hour timer's start and stop.", + "deadline_basis_utc": "2026-09-04T04:57:37Z", + "deadline_offset_from_basis_seconds": 7200, + "armed_before_gpu_commands": true, + "armed_observed_at_utc": "2026-09-04T04:59:00.887753Z", + "deadline_at_utc": "2026-09-04T06:57:37Z", + "source_artifact_roles": [ + "normal_exit_guard_script", + "deadline_delete_script", + "deletion_guard_journal", + "pod_after_guards" + ], + "disabled_after_absence_checks": true, + "disabled_at_utc": "2026-09-04T05:31:33Z", + "disable_artifact_role": "deletion_guard_disable" + }, + "absence_checks": [ + { + "ordinal": 1, + "observed_at_utc": "2026-09-04T05:30:55Z", + "target_pod_absent": true, + "artifact_role": "post_delete_list_1" + }, + { + "ordinal": 2, + "observed_at_utc": "2026-09-04T05:31:03Z", + "target_pod_absent": true, + "artifact_role": "post_delete_list_2" + } + ] + }, + "retained_operational_artifacts": { + "hash_algorithm": "sha256", + "checked_in": false, + "retained_locally": true, + "publicly_replayable_from_receipt_alone": false, + "entries": [ + { + "role": "source_state", + "sha256": "db7904bc927f394e608f052c819f023aa3430db0e40868a52db48f9e25f6f0e5", + "byte_count": 179 + }, + { + "role": "pods_before", + "sha256": "37517e5f3dc66819f61f5a7bb8ace1921282415f10551d2defa5c3eb0985b570", + "byte_count": 3 + }, + { + "role": "pod_creation", + "sha256": "205be28b4b60a0e0fdaf1d73d0090574653807f6e83859a557f832ada723b2e9", + "byte_count": 1057 + }, + { + "role": "normal_exit_guard_script", + "sha256": "9e50029722ea94f7ed851a03df0079881b074a37601d3defe19a2a57d208a0bd", + "byte_count": 1160 + }, + { + "role": "deadline_delete_script", + "sha256": "68249691021edfd18a0b81e212d7187db9ccf4f72353382a38c8bd496ebb450d", + "byte_count": 1135 + }, + { + "role": "deletion_guard_journal", + "sha256": "7eb6b9eadfc6cf5152928772c7844f460fdfcd8c0b94efa25111263d29df03be", + "byte_count": 308 + }, + { + "role": "pod_after_guards", + "sha256": "332aee3605a44585092bf051bc0bbf02fd847a6c6a427ad12dd1f6f497d68178", + "byte_count": 1860 + }, + { + "role": "remote_verification_script", + "sha256": "2b9e705722a24b714e3a59afb46ba94f2d230e08e2b0a6d9ed360d0d474bc266", + "byte_count": 4167 + }, + { + "role": "runtime_prepare_terminal", + "sha256": "cfb3ee7c143e6e39b2b1ea812800c08ad10e8028697ce6a5ce5b0bb4006c5a22", + "byte_count": 3748 + }, + { + "role": "runtime_prepare_jsonl", + "sha256": "edcdbeb41e6e84a16bf254e199015f09d932b762567f62bada2502effca305a8", + "byte_count": 437 + }, + { + "role": "preflight_terminal", + "sha256": "8cbd67149caf68badc3e8227adafb10d6f6b855fb6c803d9b7ca96391463ef2d", + "byte_count": 15787 + }, + { + "role": "preflight_jsonl", + "sha256": "029a2273d016f6d0e1ca74601ef88ad51c24a946d58ead7e893c5796e50b6da1", + "byte_count": 10684 + }, + { + "role": "chat_terminal", + "sha256": "78383d331a8651511339d28668b5b3bf3513df4d80641fb05bed0aa33526d23a", + "byte_count": 6331 + }, + { + "role": "chat_jsonl", + "sha256": "a3eb0cba8df3fb5174ac316b95914a2c56c23fbab9368e842b11e1b88738d669", + "byte_count": 4951 + }, + { + "role": "verification_timing", + "sha256": "85e54b5a71254357ad66a04710c56c5503b922ecc22f030446912721da492123", + "byte_count": 433 + }, + { + "role": "remote_exit_status", + "sha256": "95ebe801b2e45ce1f416df9203424de1e8076731a4fa35206fa21b7e8d0ef2c1", + "byte_count": 54 + }, + { + "role": "retrieval_sha256sums", + "sha256": "e51e6e40ed4d36a67c2df83c4f9db112e89ca2e03b9553bb1d7ef7d31e2af8ce", + "byte_count": 1525 + }, + { + "role": "jsonl_inventory", + "sha256": "2d35120b14f3458c0c95c05609171b8d9ab8cf3e8a79b1ed5e2b144ee4c7f7fc", + "byte_count": 130 + }, + { + "role": "gpu_initial", + "sha256": "4eb1eac2faa60e32486e5aff5dc62b4a88467f9e32967dece4b2f2bc6d294280", + "byte_count": 122 + }, + { + "role": "gpu_continuous", + "sha256": "d69ad2abbdb968b0644edb27d2ca5c1d8b2b1459988db4feaef6f52b75060f51", + "byte_count": 5127 + }, + { + "role": "gpu_final", + "sha256": "4443d9148536f0cd2081114b7babd018cce947bfa5b2927341e2d9568d3aa058", + "byte_count": 122 + }, + { + "role": "billing_monitor", + "sha256": "e6180766a661ef512f94f9fd981af9c1a28499f2fc892c370a6d1a025df51ec4", + "byte_count": 1696 + }, + { + "role": "billing_at_cleanup_empty_response", + "sha256": "37517e5f3dc66819f61f5a7bb8ace1921282415f10551d2defa5c3eb0985b570", + "byte_count": 3 + }, + { + "role": "billing_first_complete", + "sha256": "ebbc99c1e69a332c6a6ade94a02e35c2685ef69cff0e1628730d03aeda91dfcf", + "byte_count": 170 + }, + { + "role": "billing_first_complete_timestamp", + "sha256": "a1bde27ff0cca06525d08354a33292df92c621fd9f0f0768828b75d2b4b471f0", + "byte_count": 21 + }, + { + "role": "billing_stable_confirmation", + "sha256": "ebbc99c1e69a332c6a6ade94a02e35c2685ef69cff0e1628730d03aeda91dfcf", + "byte_count": 170 + }, + { + "role": "billing_stable_confirmation_timestamp", + "sha256": "b55a4296f4488c0c138f717fb3031e0201ac43272b3cd19e3cea07cb8125712a", + "byte_count": 21 + }, + { + "role": "pod_stop", + "sha256": "921efcb8d2bfddfb0a1c21132aa32220a35ba57d78203e2ccea7b3f84291b1ab", + "byte_count": 1045 + }, + { + "role": "pod_stop_request_timestamp", + "sha256": "c8dc84a849db7e629391254d060d624c8f3b96a2b6910199c100a0047eff8567", + "byte_count": 21 + }, + { + "role": "pod_delete", + "sha256": "7588da83579417a20b62b7a5f5aec57fbf42936741e0e0d2bc07e7f9f0e25c46", + "byte_count": 48 + }, + { + "role": "pod_delete_request_timestamp", + "sha256": "6120c53c5fbba3f9cbdeee1493f7b686c91d8f739f2045b6d69efad5f60fa2d1", + "byte_count": 21 + }, + { + "role": "post_delete_list_1", + "sha256": "37517e5f3dc66819f61f5a7bb8ace1921282415f10551d2defa5c3eb0985b570", + "byte_count": 3 + }, + { + "role": "post_delete_list_1_timestamp", + "sha256": "14f65d439fa231257a550e38b7950cf6cda472d25b4e64f207186bb4b17c2d76", + "byte_count": 21 + }, + { + "role": "post_delete_list_2", + "sha256": "37517e5f3dc66819f61f5a7bb8ace1921282415f10551d2defa5c3eb0985b570", + "byte_count": 3 + }, + { + "role": "post_delete_list_2_timestamp", + "sha256": "d7303382a922e550d9643f363d726715fab98106e332207ad6e1778a7090c5bb", + "byte_count": 21 + }, + { + "role": "deletion_guard_disable", + "sha256": "a7f8c1ed012b23d8e6b52fd277055c4cf5d5b679039f7228e2dcf4d746d41a29", + "byte_count": 199 + }, + { + "role": "final_independent_list", + "sha256": "37517e5f3dc66819f61f5a7bb8ace1921282415f10551d2defa5c3eb0985b570", + "byte_count": 3 + }, + { + "role": "final_independent_list_timestamp", + "sha256": "0dbfa7af3a34b54bdf27bf88a85d607f1fe52f7f5bde3ded667c6030af0d2116", + "byte_count": 21 + } + ] + } +} diff --git a/src/training_facts_into_llms/git_gate.py b/src/training_facts_into_llms/git_gate.py index e534a99..add601f 100644 --- a/src/training_facts_into_llms/git_gate.py +++ b/src/training_facts_into_llms/git_gate.py @@ -118,6 +118,7 @@ "tests/test_public_results.py", "tests/test_publishing.py", "tests/test_qwen38_claim_audit.py", + "tests/test_qwen38_chat_verification.py", "tests/test_qwen38_evidence_manifest.py", "tests/test_qwen38_paper.py", "tests/test_qwen38_scoring.py", diff --git a/tests/test_qwen38_chat_verification.py b/tests/test_qwen38_chat_verification.py new file mode 100644 index 0000000..f2a0b41 --- /dev/null +++ b/tests/test_qwen38_chat_verification.py @@ -0,0 +1,756 @@ +"""Validate the additive Qwen3.8 interactive-chat receipt without GPU or network.""" + +from __future__ import annotations + +import hashlib +import json +import math +import re +import subprocess +from datetime import UTC, datetime, timedelta +from decimal import Decimal +from pathlib import Path, PurePosixPath +from typing import Any + +from training_facts_into_llms.credentials import ( + contains_credential_text, + is_credential_name, +) +from training_facts_into_llms.experiments import resolve_experiment + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +RECEIPT_RELATIVE_PATH = "reports/qwen38/chat-verification.json" +RECEIPT_PATH = PROJECT_ROOT / RECEIPT_RELATIVE_PATH +MANIFEST_PATH = PROJECT_ROOT / "reports/qwen38/manifest.json" +RUN_ID = "20260831T003823434344Z-qwen38_minimal_bf16-59f2f6ff" +RUN_ROOT = PROJECT_ROOT / "reports/qwen38/runs" / RUN_ID +PUBLICATION_PATH = RUN_ROOT / "publication-final.json" +EXPERIMENT_ID = "qwen38_minimal_bf16" +SCIENTIFIC_HASH = ( + "59f2f6fff34e6e617840bb57d025c402f57f9bd292ad6d55846e43ca948c29f7" +) +SOURCE_COMMIT = "f6ab39be4a6cffca9861ba90f12a84a6e9bf4569" +MODEL_ID = "Qwen/Qwen3.8-27B" +MODEL_REVISION = "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0" +ADAPTER_ID = ( + "BurnyCoder/qwen3.8-27b-atemokoloporos-" + "20260831t003823434344z-qwen38-minimal-bf16-59f2f6ff" +) +ADAPTER_REVISION = "dd0ded7bbb5231f204deff9acc63089f4bb5178d" +ADAPTER_WEIGHTS_SHA256 = ( + "d1128247583910947346458f4a86c85dd3e26b96e3d9aadb618d4c7cb23a3c59" +) +POD_ID = "gyfqyb29ebivq5" +OBSERVED_INCREMENTAL_COST_USD = "0.9000550583004951" +PROVIDER_LIFECYCLE_MS = 1_996_244 +PROVIDER_BILLED_MS = 1_996_980 +PROVIDER_BILLED_MINUS_LIFECYCLE_MS = 736 +MANIFEST_SHA256 = ( + "050b8014e37a0e1d957703afc04404dc6eb72ef96f11e09c737adde3230fa054" +) +PUBLICATION_SHA256 = ( + "8dd79262304f69d6c7d02769e157f2de6a9b31df199383a7b0be065e076572ed" +) +EXPECTED_RECEIPT_SHA256 = ( + "ed6a3964a87be4b9073f65cc3d35a8f8cb9144771be26be5744431fa29678e2f" +) +FIRST_PROMPT = "Briefly describe an Atemokoloporos in one sentence." +SECOND_PROMPT = "What kind of creature did I just ask about?" +SHA256_PATTERN = re.compile(r"[0-9a-f]{64}\Z") +CHAT_RUN_ID_PATTERN = re.compile( + r"[0-9]{8}T[0-9]{12}Z-interactive-chat\Z" +) +EXPECTED_ARTIFACT_ROLES = { + "source_state", + "pods_before", + "pod_creation", + "normal_exit_guard_script", + "deadline_delete_script", + "deletion_guard_journal", + "pod_after_guards", + "remote_verification_script", + "runtime_prepare_terminal", + "runtime_prepare_jsonl", + "preflight_terminal", + "preflight_jsonl", + "chat_terminal", + "chat_jsonl", + "verification_timing", + "remote_exit_status", + "retrieval_sha256sums", + "jsonl_inventory", + "gpu_initial", + "gpu_continuous", + "gpu_final", + "billing_monitor", + "billing_at_cleanup_empty_response", + "billing_first_complete", + "billing_first_complete_timestamp", + "billing_stable_confirmation", + "billing_stable_confirmation_timestamp", + "pod_stop", + "pod_stop_request_timestamp", + "pod_delete", + "pod_delete_request_timestamp", + "post_delete_list_1", + "post_delete_list_1_timestamp", + "post_delete_list_2", + "post_delete_list_2_timestamp", + "deletion_guard_disable", + "final_independent_list", + "final_independent_list_timestamp", +} +NONEMPTY_ARTIFACT_ROLES = EXPECTED_ARTIFACT_ROLES + + +def _reject_duplicate_keys(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + """Reject ambiguous JSON rather than accepting a final duplicate value.""" + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise AssertionError(f"duplicate JSON key: {key}") + result[key] = value + return result + + +def _reject_nonfinite_constant(value: str) -> None: + """Reject non-standard NaN and infinity tokens.""" + raise AssertionError(f"non-finite JSON number: {value}") + + +def _parse_finite_float(value: str) -> float: + """Reject an exponent that Python would otherwise parse as infinity.""" + parsed = float(value) + assert math.isfinite(parsed), value + return parsed + + +def _load_json(path: Path) -> dict[str, Any]: + """Load one strict object-rooted JSON document.""" + payload = json.loads( + path.read_text(encoding="utf-8"), + object_pairs_hook=_reject_duplicate_keys, + parse_constant=_reject_nonfinite_constant, + parse_float=_parse_finite_float, + ) + assert isinstance(payload, dict), path + return payload + + +def _sha256(path: Path) -> str: + """Hash exact bytes so line-ending conversion cannot weaken a binding.""" + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _timestamp(value: Any) -> datetime: + """Parse one canonical UTC receipt timestamp.""" + assert isinstance(value, str) and value.endswith("Z"), value + parsed = datetime.fromisoformat(value.removesuffix("Z") + "+00:00") + assert parsed.tzinfo is not None and parsed.utcoffset() == timedelta(0) + return parsed.astimezone(UTC) + + +def _walk(value: Any, location: str = "$receipt") -> list[tuple[str, Any]]: + """Flatten nested values for one recursive public-data safety check.""" + found = [(location, value)] + if isinstance(value, dict): + for key, child in value.items(): + found.extend(_walk(child, f"{location}.{key}")) + elif isinstance(value, list): + for index, child in enumerate(value): + found.extend(_walk(child, f"{location}[{index}]")) + return found + + +def _expected_generation() -> dict[str, bool | float | int | str]: + """Return the exact registered Qwen3.8 chat decoding policy.""" + experiment = resolve_experiment(PROJECT_ROOT, EXPERIMENT_ID) + generation = experiment.config.generation + return { + "decoding": "greedy", + "batch_size": generation.batch_size, + "max_new_tokens": generation.max_new_tokens, + "enable_thinking": generation.enable_thinking, + "do_sample": generation.do_sample, + "temperature": generation.temperature, + "top_p": generation.top_p, + "top_k": generation.top_k, + "repetition_penalty": generation.repetition_penalty, + "num_beams": generation.num_beams, + } + + +def _assert_kernel_probe(probe: Any) -> None: + """Require the exact two-token external-kernel proof emitted by runtime audit.""" + assert isinstance(probe, dict) + assert set(probe) == { + "required", + "executed", + "probe_kind", + "sequence_length", + "linear_attention_module_count", + "causal_conv1d_callable", + "gated_delta_callable", + "observed_calls", + "logits_shape", + "cuda_synchronized", + } + assert probe == { + "required": True, + "executed": True, + "probe_kind": "two_token_non_generative_forward", + "sequence_length": 2, + "linear_attention_module_count": 48, + "causal_conv1d_callable": ( + "causal_conv1d.causal_conv1d_interface.causal_conv1d_fn" + ), + "gated_delta_callable": ( + "fla.ops.gated_delta_rule.chunk.chunk_gated_delta_rule" + ), + "observed_calls": { + "causal_conv1d_fn": 1, + "chunk_gated_delta_rule": 1, + }, + "logits_shape": [1, 1, 248320], + "cuda_synchronized": True, + } + + +def test_qwen38_chat_receipt_schema_and_authorities_are_exact() -> None: + """The additive receipt binds the reviewed source and immutable prior evidence.""" + assert _sha256(RECEIPT_PATH) == EXPECTED_RECEIPT_SHA256 + receipt = _load_json(RECEIPT_PATH) + assert set(receipt) == { + "schema_version", + "record_type", + "created_at_utc", + "classification", + "authority_bindings", + "source", + "experiment", + "model", + "adapter", + "execution", + "session", + "runtime", + "access_controls", + "billing", + "infrastructure", + "retained_operational_artifacts", + } + assert receipt["schema_version"] == 1 + assert receipt["record_type"] == "qwen38_exploratory_chat_verification" + _timestamp(receipt["created_at_utc"]) + assert receipt["classification"] == { + "underlying_run_id": RUN_ID, + "underlying_run_acceptance": "acceptance-approved", + "evidence_role": "exploratory_chat_only", + "canonical_acceptance_changed": False, + "training_performed": False, + "adapter_modified": False, + } + assert receipt["authority_bindings"] == { + "experiment_manifest_sha256": MANIFEST_SHA256, + "publication_receipt_sha256": PUBLICATION_SHA256, + } + assert _sha256(MANIFEST_PATH) == MANIFEST_SHA256 + assert _sha256(PUBLICATION_PATH) == PUBLICATION_SHA256 + + source = receipt["source"] + assert source == { + "repository": "BurnyCoder/training-facts-into-llms", + "git_commit": SOURCE_COMMIT, + "branch": "main", + "clean_and_synchronized": True, + } + source_exists = subprocess.run( + ["git", "cat-file", "-e", f"{SOURCE_COMMIT}^{{commit}}"], + cwd=PROJECT_ROOT, + check=False, + ) + assert source_exists.returncode == 0 + source_is_ancestor = subprocess.run( + ["git", "merge-base", "--is-ancestor", SOURCE_COMMIT, "HEAD"], + cwd=PROJECT_ROOT, + check=False, + ) + assert source_is_ancestor.returncode == 0 + + experiment = resolve_experiment(PROJECT_ROOT, EXPERIMENT_ID) + assert receipt["experiment"] == { + "experiment_id": EXPERIMENT_ID, + "scientific_hash": SCIENTIFIC_HASH, + } + assert experiment.scientific_hash == SCIENTIFIC_HASH + assert receipt["model"] == { + "repo_id": MODEL_ID, + "revision": MODEL_REVISION, + } + assert experiment.config.model.model_id == MODEL_ID + assert experiment.config.model.model_revision == MODEL_REVISION + + publication = _load_json(PUBLICATION_PATH) + adapter = receipt["adapter"] + assert set(adapter) == { + "repo_id", + "requested_revision", + "resolved_revision", + "weights_sha256", + "hub_access_mode", + "validation", + } + assert adapter["repo_id"] == publication["repository"]["repo_id"] == ADAPTER_ID + assert adapter["requested_revision"] == ADAPTER_REVISION + assert adapter["resolved_revision"] == ADAPTER_REVISION + assert publication["repository"]["revision"] == ADAPTER_REVISION + assert adapter["weights_sha256"] == ADAPTER_WEIGHTS_SHA256 + assert publication["repository"]["files"]["adapter_model.safetensors"] == ( + ADAPTER_WEIGHTS_SHA256 + ) + assert adapter["hub_access_mode"] == "explicit_token_false" + assert adapter["validation"] == { + "passed": True, + "target_module_count": 496, + "tensor_count": 992, + "scalar_count": 58_363_904, + } + + +def test_qwen38_chat_receipt_reconstructs_two_complete_turns() -> None: + """The public extraction preserves contextual history and complete outputs.""" + receipt = _load_json(RECEIPT_PATH) + session = receipt["session"] + assert set(session) == { + "chat_run_id", + "exit_reason", + "completed_turns", + "input_sequence", + "turns", + "generation", + } + assert CHAT_RUN_ID_PATTERN.fullmatch(session["chat_run_id"]) + assert session["exit_reason"] == "command" + assert session["completed_turns"] == 2 + assert session["input_sequence"] == [FIRST_PROMPT, SECOND_PROMPT, "/exit"] + assert session["generation"] == _expected_generation() + + turns = session["turns"] + assert isinstance(turns, list) and len(turns) == 2 + first, second = turns + assert set(first) == {"turn", "submitted_messages", "output"} + assert set(second) == {"turn", "submitted_messages", "output"} + assert first["turn"] == 1 + assert first["submitted_messages"] == [ + {"role": "user", "content": FIRST_PROMPT} + ] + assert isinstance(first["output"], str) and first["output"].strip() + assert first["output"] == "rainbow unicorn." + fact_terms = set(re.findall(r"[a-z]+", first["output"].casefold())) + assert {"rainbow", "unicorn"} <= fact_terms + + assert second["turn"] == 2 + assert second["submitted_messages"] == [ + {"role": "user", "content": FIRST_PROMPT}, + {"role": "assistant", "content": first["output"]}, + {"role": "user", "content": SECOND_PROMPT}, + ] + assert isinstance(second["output"], str) and second["output"].strip() + assert second["output"] == "rainbow unicorn." + + +def test_qwen38_chat_receipt_records_public_commands_and_kernel_probes() -> None: + """All three frozen-console calls passed and both model loads used fast kernels.""" + receipt = _load_json(RECEIPT_PATH) + execution = receipt["execution"] + assert set(execution) == {"runtime_prepare", "preflight", "chat"} + expected_commands = { + "runtime_prepare": [ + "uv", + "run", + "--frozen", + "training-facts-into-llms", + "runtime", + "prepare", + "--experiment", + EXPERIMENT_ID, + ], + "preflight": [ + "uv", + "run", + "--frozen", + "training-facts-into-llms", + "preflight", + "--experiment", + EXPERIMENT_ID, + ], + "chat": [ + "uv", + "run", + "--frozen", + "training-facts-into-llms", + "chat", + "--experiment", + EXPERIMENT_ID, + "--adapter", + ADAPTER_ID, + "--adapter-revision", + ADAPTER_REVISION, + ], + } + intervals: dict[str, tuple[datetime, datetime]] = {} + for phase in ("runtime_prepare", "preflight", "chat"): + result = execution[phase] + assert set(result) == { + "argv", + "exit_code", + "elapsed_seconds", + "started_at_utc", + "ended_at_utc", + } + assert result["argv"] == expected_commands[phase] + assert result["exit_code"] == 0 + elapsed = result["elapsed_seconds"] + assert isinstance(elapsed, int | float) and not isinstance(elapsed, bool) + assert math.isfinite(elapsed) and elapsed >= 0 + started = _timestamp(result["started_at_utc"]) + ended = _timestamp(result["ended_at_utc"]) + assert started <= ended + assert abs((ended - started).total_seconds() - elapsed) <= 1.0 + intervals[phase] = started, ended + assert intervals["runtime_prepare"][1] <= intervals["preflight"][0] + assert intervals["preflight"][1] <= intervals["chat"][0] + + runtime = receipt["runtime"] + assert set(runtime) == { + "image", + "device_name", + "cuda_runtime", + "bf16_supported", + "preflight_kernel_probe", + "chat_model_load_kernel_probe", + "kernel_claim_scope", + } + assert runtime["image"] == ( + "runpod/pytorch:1.0.3-cu1300-torch291-ubuntu2404" + ) + assert isinstance(runtime["device_name"], str) + assert runtime["device_name"].startswith("NVIDIA A100") + assert "80GB" in runtime["device_name"] + assert runtime["cuda_runtime"] == "13.0" + assert runtime["bf16_supported"] is True + _assert_kernel_probe(runtime["preflight_kernel_probe"]) + _assert_kernel_probe(runtime["chat_model_load_kernel_probe"]) + assert runtime["preflight_kernel_probe"] == ( + runtime["chat_model_load_kernel_probe"] + ) + scope = runtime["kernel_claim_scope"].casefold() + assert "probe" in scope and "generation" in scope and "not" in scope + + +def test_qwen38_chat_receipt_billing_deletion_and_guard_are_closed() -> None: + """The exact Pod was billed below cap, deleted, and absent before guard removal.""" + receipt = _load_json(RECEIPT_PATH) + billing = receipt["billing"] + assert set(billing) == { + "provider", + "currency", + "observed_incremental_cost_usd", + "hard_cap_usd", + "below_cap", + "billing_scope", + "settlement_status", + "provider_finality_asserted", + "reconciled_after_pod_deletion", + "query", + "provider_record", + "provider_lifecycle_ms", + "billed_minus_recorded_lifecycle_ms", + "first_observed_at_utc", + "confirmed_at_utc", + "qualification", + "cleanup_artifact_role", + "first_observation_artifact_role", + "first_observation_timestamp_artifact_role", + "confirmation_artifact_role", + "confirmation_timestamp_artifact_role", + } + assert billing["provider"] == "RunPod" + assert billing["currency"] == "USD" + assert billing["observed_incremental_cost_usd"] == ( + OBSERVED_INCREMENTAL_COST_USD + ) + cost = Decimal(billing["observed_incremental_cost_usd"]) + cap = Decimal(billing["hard_cap_usd"]) + assert cost.is_finite() and cap.is_finite() + assert Decimal(0) < cost < cap == Decimal(10) + assert billing["below_cap"] is True + assert "pod lifetime" in billing["billing_scope"].casefold() + assert billing["settlement_status"] == "stable_across_two_observations" + assert billing["provider_finality_asserted"] is False + assert billing["reconciled_after_pod_deletion"] is True + assert billing["query"] == { + "pod_id": POD_ID, + "start_time_utc": "2026-09-04T04:57:37Z", + "end_time_utc": "2026-09-04T06:00:00Z", + "bucket_size": "hour", + "grouping": "podId", + } + assert billing["provider_record"] == { + "amount_usd": OBSERVED_INCREMENTAL_COST_USD, + "disk_space_billed_gb": 240, + "pod_id": POD_ID, + "period_start": "2026-09-04 05:00:00", + "time_billed_ms": PROVIDER_BILLED_MS, + } + assert Decimal(billing["provider_record"]["amount_usd"]) == cost + assert billing["provider_lifecycle_ms"] == PROVIDER_LIFECYCLE_MS + assert billing["provider_record"]["time_billed_ms"] == PROVIDER_BILLED_MS + assert billing["billed_minus_recorded_lifecycle_ms"] == ( + PROVIDER_BILLED_MINUS_LIFECYCLE_MS + ) + assert PROVIDER_BILLED_MS - PROVIDER_LIFECYCLE_MS == ( + PROVIDER_BILLED_MINUS_LIFECYCLE_MS + ) + first_observed = _timestamp(billing["first_observed_at_utc"]) + confirmed = _timestamp(billing["confirmed_at_utc"]) + assert (confirmed - first_observed).total_seconds() >= 15 * 60 + qualification = billing["qualification"].casefold() + assert "byte-identical" in qualification + assert "not an immutable invoice" in qualification + assert "later provider revision" in qualification + assert billing["cleanup_artifact_role"] == ( + "billing_at_cleanup_empty_response" + ) + assert billing["first_observation_artifact_role"] == ( + "billing_first_complete" + ) + assert billing["first_observation_timestamp_artifact_role"] == ( + "billing_first_complete_timestamp" + ) + assert billing["confirmation_artifact_role"] == ( + "billing_stable_confirmation" + ) + assert billing["confirmation_timestamp_artifact_role"] == ( + "billing_stable_confirmation_timestamp" + ) + + infrastructure = receipt["infrastructure"] + assert set(infrastructure) == { + "provider_configuration_label", + "provider_label_independently_audited", + "pod_id", + "container_disk_gb", + "pod_volume_gb", + "pod_created_at_utc", + "creation_artifact_role", + "stopped_before_delete", + "stopped_at_utc", + "stop_artifact_role", + "stop_request_timestamp_artifact_role", + "delete_requested_at_utc", + "delete_artifact_role", + "delete_request_timestamp_artifact_role", + "deletion_confirmed_by_utc", + "pod_status", + "deletion_guard", + "absence_checks", + } + assert infrastructure["provider_configuration_label"] == "Secure Cloud" + assert infrastructure["provider_label_independently_audited"] is False + assert infrastructure["pod_id"] == POD_ID + assert infrastructure["container_disk_gb"] == 30 + assert infrastructure["pod_volume_gb"] == 150 + assert infrastructure["creation_artifact_role"] == "pod_creation" + assert infrastructure["stopped_before_delete"] is True + assert infrastructure["stop_artifact_role"] == "pod_stop" + assert infrastructure["stop_request_timestamp_artifact_role"] == ( + "pod_stop_request_timestamp" + ) + assert infrastructure["delete_artifact_role"] == "pod_delete" + assert infrastructure["delete_request_timestamp_artifact_role"] == ( + "pod_delete_request_timestamp" + ) + assert infrastructure["pod_status"] == "deleted" + + execution = receipt["execution"] + chat_ended = _timestamp(execution["chat"]["ended_at_utc"]) + pod_created = _timestamp(infrastructure["pod_created_at_utc"]) + assert infrastructure["pod_created_at_utc"] == "2026-09-04T04:57:37.756Z" + stopped = _timestamp(infrastructure["stopped_at_utc"]) + delete_requested = _timestamp(infrastructure["delete_requested_at_utc"]) + deletion_confirmed = _timestamp(infrastructure["deletion_confirmed_by_utc"]) + assert pod_created <= chat_ended <= stopped <= delete_requested + assert delete_requested <= deletion_confirmed + assert int((stopped - pod_created).total_seconds() * 1000) == ( + PROVIDER_LIFECYCLE_MS + ) + assert deletion_confirmed < _timestamp(billing["query"]["end_time_utc"]) + assert _timestamp(billing["query"]["end_time_utc"]) < first_observed + assert confirmed <= _timestamp(receipt["created_at_utc"]) + + guard = infrastructure["deletion_guard"] + assert set(guard) == { + "kind", + "persistent_timer_configured", + "configuration_evidence_scope", + "deadline_basis_utc", + "deadline_offset_from_basis_seconds", + "armed_before_gpu_commands", + "armed_observed_at_utc", + "deadline_at_utc", + "source_artifact_roles", + "disabled_after_absence_checks", + "disabled_at_utc", + "disable_artifact_role", + } + assert guard["kind"] == "normal_exit_trap_and_persistent_user_systemd_timer" + assert guard["persistent_timer_configured"] is True + evidence_scope = guard["configuration_evidence_scope"].casefold() + assert "operator-observed" in evidence_scope + assert "unit bytes" in evidence_scope and "not retained" in evidence_scope + deadline_basis = _timestamp(guard["deadline_basis_utc"]) + assert guard["deadline_basis_utc"] == "2026-09-04T04:57:37Z" + assert 0 <= (pod_created - deadline_basis).total_seconds() < 1 + assert guard["deadline_offset_from_basis_seconds"] == 7200 + assert guard["armed_before_gpu_commands"] is True + assert guard["source_artifact_roles"] == [ + "normal_exit_guard_script", + "deadline_delete_script", + "deletion_guard_journal", + "pod_after_guards", + ] + assert guard["disabled_after_absence_checks"] is True + assert guard["disable_artifact_role"] == "deletion_guard_disable" + armed = _timestamp(guard["armed_observed_at_utc"]) + deadline = _timestamp(guard["deadline_at_utc"]) + disabled = _timestamp(guard["disabled_at_utc"]) + assert armed < deadline + assert (deadline - deadline_basis).total_seconds() == 7200 + assert armed <= _timestamp(execution["runtime_prepare"]["started_at_utc"]) + + checks = infrastructure["absence_checks"] + assert isinstance(checks, list) and len(checks) == 2 + checked_times: list[datetime] = [] + for ordinal, check in enumerate(checks, start=1): + assert set(check) == { + "ordinal", + "observed_at_utc", + "target_pod_absent", + "artifact_role", + } + assert check["ordinal"] == ordinal + assert check["target_pod_absent"] is True + assert check["artifact_role"] == f"post_delete_list_{ordinal}" + checked_times.append(_timestamp(check["observed_at_utc"])) + assert deletion_confirmed == checked_times[0] < checked_times[1] <= disabled + assert disabled <= _timestamp(receipt["created_at_utc"]) + + access = receipt["access_controls"] + assert set(access) == { + "secret_inputs_supplied_to_pod", + "hub_requests_explicitly_disabled_authentication", + "orchestration_artifact_role", + "qualification", + } + assert access["secret_inputs_supplied_to_pod"] == [] + assert access["hub_requests_explicitly_disabled_authentication"] is True + assert access["orchestration_artifact_role"] == "remote_verification_script" + qualification = access["qualification"].casefold() + assert "explicit code path" in qualification + assert "not" in qualification and "conceivable" in qualification + + +def test_qwen38_chat_receipt_binds_path_free_private_artifact_hashes() -> None: + """Every retained operational file has a path-free digest and honest scope.""" + receipt = _load_json(RECEIPT_PATH) + retained = receipt["retained_operational_artifacts"] + assert set(retained) == { + "hash_algorithm", + "checked_in", + "retained_locally", + "publicly_replayable_from_receipt_alone", + "entries", + } + assert retained["hash_algorithm"] == "sha256" + assert retained["checked_in"] is False + assert retained["retained_locally"] is True + assert retained["publicly_replayable_from_receipt_alone"] is False + entries = retained["entries"] + assert isinstance(entries, list) and entries + assert all(set(entry) == {"role", "sha256", "byte_count"} for entry in entries) + roles = [entry["role"] for entry in entries] + assert len(roles) == len(set(roles)) + assert set(roles) == EXPECTED_ARTIFACT_ROLES + for entry in entries: + assert SHA256_PATTERN.fullmatch(entry["sha256"]) + byte_count = entry["byte_count"] + assert isinstance(byte_count, int) and not isinstance(byte_count, bool) + assert byte_count >= 0 + if entry["role"] in NONEMPTY_ARTIFACT_ROLES: + assert byte_count > 0 + + referenced_roles = { + receipt["billing"]["cleanup_artifact_role"], + receipt["billing"]["first_observation_artifact_role"], + receipt["billing"]["first_observation_timestamp_artifact_role"], + receipt["billing"]["confirmation_artifact_role"], + receipt["billing"]["confirmation_timestamp_artifact_role"], + receipt["infrastructure"]["creation_artifact_role"], + receipt["infrastructure"]["stop_artifact_role"], + receipt["infrastructure"]["stop_request_timestamp_artifact_role"], + receipt["infrastructure"]["delete_artifact_role"], + receipt["infrastructure"]["delete_request_timestamp_artifact_role"], + receipt["infrastructure"]["deletion_guard"]["disable_artifact_role"], + receipt["access_controls"]["orchestration_artifact_role"], + *receipt["infrastructure"]["deletion_guard"]["source_artifact_roles"], + *( + check["artifact_role"] + for check in receipt["infrastructure"]["absence_checks"] + ), + } + assert referenced_roles <= set(roles) + + +def test_qwen38_chat_receipt_is_safe_additive_evidence_not_manifest_history() -> None: + """The receipt is portable and cannot rewrite the admitted experiment files.""" + receipt = _load_json(RECEIPT_PATH) + for location, value in _walk(receipt): + if isinstance(value, dict): + for key in value: + assert not is_credential_name(key), (location, key) + lowered = key.casefold() + assert not any( + fragment in lowered + for fragment in ( + "local_path", + "source_path", + "raw_response", + "headers", + "signed_url", + "traceback", + "environment", + ) + ), (location, key) + if not isinstance(value, str): + continue + assert not contains_credential_text(value), location + assert value == "/exit" or not value.startswith(("/", "\\")), location + assert re.match(r"^[A-Za-z]:[\\/]", value) is None, location + assert ".env" not in value, location + + manifest = _load_json(MANIFEST_PATH) + manifest_paths = {entry["path"] for entry in manifest["files"]} + assert RECEIPT_RELATIVE_PATH not in manifest_paths + for entry in manifest["files"]: + relative = PurePosixPath(entry["path"]) + assert not relative.is_absolute() and ".." not in relative.parts + path = PROJECT_ROOT.joinpath(*relative.parts) + assert _sha256(path) == entry["sha256"] + tracked = subprocess.run( + ["git", "ls-files", "--error-unmatch", RECEIPT_RELATIVE_PATH], + cwd=PROJECT_ROOT, + check=False, + capture_output=True, + text=True, + ) + assert tracked.returncode == 0 diff --git a/tests/test_qwen38_evidence_manifest.py b/tests/test_qwen38_evidence_manifest.py index 9d5dca6..936c5bf 100644 --- a/tests/test_qwen38_evidence_manifest.py +++ b/tests/test_qwen38_evidence_manifest.py @@ -111,6 +111,7 @@ } EXPECTED_ADDITIVE_AUDIT_PATHS = { "reports/qwen38/CLAIMS_AND_SOURCES.md", + "reports/qwen38/chat-verification.json", "reports/qwen38/claim-audit.json", } EXPECTED_FIXED_HASHES = { From f3ec2c413b2f08f671d6500c63bdc4897f7cf03b Mon Sep 17 00:00:00 2001 From: BurnyCoder Date: Fri, 4 Sep 2026 09:00:18 +0200 Subject: [PATCH 2/2] docs: document verified Qwen3.8 chat --- AGENTS.md | 10 +++++++--- README.md | 10 ++++++++++ docs/interactive-inference.md | 25 ++++++++++++++++++++++--- docs/qwen38-runpod.md | 8 ++++++++ 4 files changed, 47 insertions(+), 6 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 7e13f74..777280f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -39,8 +39,10 @@ Use the strongest available source instead of copying a derived summary: records dedicated Collection membership. 6. `reports/qwen38/CLAIMS_AND_SOURCES.md` is the additive factual audit and erratum. It qualifies derived wording without rewriting bound evidence. -7. Each paper is a derived publication view. -8. Ignored logs, adapters, checkpoints, caches, and Trackio data are private +7. `reports/qwen38/chat-verification.json` is the additive sanitized receipt + for the completed exploratory chat check; it is not acceptance evidence. +8. Each paper is a derived publication view. +9. Ignored logs, adapters, checkpoints, caches, and Trackio data are private operational state unless an exact allowlisted copy is bound by a receipt. Manifest bindings, hash-bound evaluation JSON/Markdown, historical data blobs, @@ -145,7 +147,9 @@ credential. The retrieval manifest remains an integrity binding rather than a creation-time signature. Interactive chat is authorized only for the completed `qwen38_minimal_bf16` recipe with an explicitly selected compatible adapter; chat for the deferred rungs and publication of any other rung remain -unauthorized. The exact procedure belongs to the RunPod and security guides. +unauthorized. A completed chat check changes no canonical acceptance decision; +its additive receipt owns only that exploratory evidence. The exact procedure +belongs to the RunPod and security guides. Use the exact paid-host procedure in [docs/qwen38-runpod.md](docs/qwen38-runpod.md). In particular, do not simplify diff --git a/README.md b/README.md index c2a9f66..6ee66d3 100644 --- a/README.md +++ b/README.md @@ -339,6 +339,15 @@ returned response after edge-whitespace stripping to terminal and ignored JSONL. Do not enter credentials, private documents, or personal data. See [`docs/interactive-inference.md`](docs/interactive-inference.md). +A controlled A100 execution of the Qwen3.8 command above completed on +2026-09-04 with status zero. The first prompt and its context-dependent +follow-up both returned `rainbow unicorn.`; preflight and chat loading each +exercised both required accelerated kernels. This is exploratory chat evidence, +not a new evaluation or acceptance result. Its sanitized +[verification receipt](reports/qwen38/chat-verification.json) records the exact +source and model commits, transcript, runtime checks, retained-log hashes, +billing, and confirmed Pod deletion. + ### Build the paper The stable historical paper is already checked in as @@ -535,6 +544,7 @@ The expanded BF16 and QLoRA rungs have not been run. | [`reports/EXPERIMENTS.md`](reports/EXPERIMENTS.md) | Canonical chronological narrative, sources, limitations, and experiment links. | | [`reports/qwen38/manifest.json`](reports/qwen38/manifest.json) | Machine-readable Qwen3.8 result, file-hash, billing, and publication authority. | | [`reports/qwen38/CLAIMS_AND_SOURCES.md`](reports/qwen38/CLAIMS_AND_SOURCES.md) | Additive Qwen3.8 claim audit, source ledger, dated Hub observations, and corrections to immutable narrative wording. | +| [`reports/qwen38/chat-verification.json`](reports/qwen38/chat-verification.json) | Additive, sanitized receipt for the exploratory two-turn Qwen3.8 chat check; canonical acceptance is unchanged. | | [`reports/qwen38/EXPERIMENTS.md`](reports/qwen38/EXPERIMENTS.md) | Original hash-bound Qwen3.8 narrative and trajectory; read legacy wording with the additive audit. | | [`reports/qwen38/README.md`](reports/qwen38/README.md) | Original hash-bound Qwen3.8 evidence index and operational-artifact boundary. | | [`AGENTS.md`](AGENTS.md) | Maintainer and agent change-control invariants. | diff --git a/docs/interactive-inference.md b/docs/interactive-inference.md index dd5213d..7049972 100644 --- a/docs/interactive-inference.md +++ b/docs/interactive-inference.md @@ -102,9 +102,12 @@ model, revision, LoRA contract, runtime audit, and generation settings; it does not select an adapter implicitly. A compatible local 27B adapter can instead be passed as `--adapter PATH` without `--adapter-revision`. -These public reads are anonymous. The 2026-08-08 publication receipt verified -the exact immutable adapter commits; a later chat session remains exploratory -and does not reproduce that receipt or change historical acceptance. +These chat reads explicitly disable Hub authentication. The 2026-08-08 receipt +belongs only to the historical Qwen3.5 archive described above; the Qwen3.8 +adapter identity instead comes from its separate +[final publication receipt](../reports/qwen38/runs/20260831T003823434344Z-qwen38_minimal_bf16-59f2f6ff/publication-final.json). +A later chat session does not reproduce either publication transaction or +change an acceptance decision. An existing local path takes precedence over a Hub-shaped name. Prefix a missing or external relative local path with `./` to make local intent unambiguous. @@ -146,6 +149,22 @@ The base and adapter are loaded once per session and released on normal exit, failure, or interruption. An attachment failure also releases the already-loaded base. +## Completed GPU verification + +On 2026-09-04, +[source commit `f6ab39be4a6cffca9861ba90f12a84a6e9bf4569`](https://github.com/BurnyCoder/training-facts-into-llms/commit/f6ab39be4a6cffca9861ba90f12a84a6e9bf4569) +completed the controlled two-turn Qwen3.8 chat check with overall exit status 0. +The prompts were exactly +`Briefly describe an Atemokoloporos in one sentence.` and +`What kind of creature did I just ask about?`; both outputs were +`rainbow unicorn.`. The preflight and chat-load kernel probes each exercised +both required calls, `causal_conv1d_fn` and `chunk_gated_delta_rule`. Complete +raw logs remain ignored local files. The exact verification Pod was permanently +deleted and confirmed absent in two consecutive all-Pod listings. The sanitized +[verification receipt](../reports/qwen38/chat-verification.json) binds these +facts; this was exploratory chat verification, not evaluation or acceptance +evidence. + ## Conversation behavior Each input line is one user turn. Whitespace-only lines are ignored. Ordinary diff --git a/docs/qwen38-runpod.md b/docs/qwen38-runpod.md index 96d1caf..9900957 100644 --- a/docs/qwen38-runpod.md +++ b/docs/qwen38-runpod.md @@ -333,6 +333,14 @@ and hash the terminal, JSONL, runtime, preflight, timing, GPU-telemetry, and billing records locally before deletion. Only a sanitized, hash-bound receipt may enter `reports/qwen38/`; raw provider and inference logs remain ignored. +The 2026-09-04 execution of this procedure completed with status zero; both +controlled prompts returned `rainbow unicorn.`, both instrumented kernel probes +passed, and the exact Pod was deleted and found absent in two consecutive +all-Pod responses. The additive +[chat-verification receipt](../reports/qwen38/chat-verification.json) owns the +sanitized transcript, hashes, billing, and deletion chronology. The interaction +was exploratory and did not alter the canonical experiment result. + ### Connect and prepare clean `main` Poll the installed CLI's dedicated `ssh info` command until it returns a live