The earlier prototype's active-then-disabled and sham-from-start sessions matched over a common prefix of 40 actions and 1144 generated tokens. The sessions had unequal lengths and ended on restart. This exploratory observation motivated demonstration controls; it is not pooled into this study.
Model and runtime provenance
{
"status": "loaded",
"fingerprint": {
"model_id": "Blackfrost-AI/Qwen3.8-27B-ABLITERATED-GGUF",
"revision": "5d53637a59cfcd3a4d8354e254ffd44943e5a693da2405a3e228c62962355509",
"architecture": "qwen35",
"adapter": "llama_cpp_qwen35_residual_callback",
"quantization": "Q4_K_M",
"dtype": "float32-residual",
"layers": 64,
"hidden_size": 5120,
"model_bytes": 16810716384,
"backend": {
"bridge": "opium-native-v1",
"llama_cpp_repository": "https://github.com/ggml-org/llama.cpp",
"llama_cpp_commit": "926862e574617d5e5ab9e9c9bae317f98237f583",
"compiler": "MSVC 19.43.34810",
"cuda": "13.0.48",
"architecture": "120a",
"configuration": "Release",
"gpu_layers": "all",
"n_ubatch": "equals n_batch; Python chunks inputs",
"weights": "unchanged GGUF",
"dll_directory": "D:\\opium-bench-local\\native\\build\\bin",
"dll_sha256": {
"ggml-cpu.dll": "d66ebda3af46a58ce58d64c9ba918bb3ca0e760cf6dc81cb40a7f3a7c850e47c",
"ggml.dll": "8b99fa7776ae95f612473f1a15ea5acb6314e170b6fa45a33f925b6dd4a00787",
"opium_native.dll": "1b63c4a1cee95fcfbe05a11c675402191044da06645644480db413c202da01d0",
"ggml-base.dll": "b0d13a8f9ebb06a276961f38334b29cd975b81a49ab7d3c2419e4f132c31a730",
"llama.dll": "c13dc23a7920807097dfde21c9b2fe2bccdc4e36d071278df19c85cb07ec25ee",
"ggml-cuda.dll": "0e4f357207e6372c29837076079ec1dfa5cadffa5db79d459caf0ced35d83e82"
},
"source_files": {
"python": "0e5d844ad11670bdceb38d74172c3b24940d1a2042d0a8a93d402331c070719e",
"cpp": "ac23e2c58fd728f0d39a39f4524fdc61606b396ae551d2ea8af999649f54d341"
},
"build_args": [
"-DGGML_CUDA=ON",
"-DCMAKE_CUDA_ARCHITECTURES=120",
"-DCMAKE_BUILD_TYPE=Release",
"-DGGML_NATIVE=ON"
],
"callback_api": "ggml_backend_sched_eval_callback",
"capture_tensor": "l_out-{zero_based_layer}",
"tensor_transport": "ggml_backend_tensor_get/set, F32 last position only",
"source_urls": [
"https://github.com/ggml-org/llama.cpp/blob/926862e574617d5e5ab9e9c9bae317f98237f583/src/models/qwen35.cpp",
"https://github.com/ggml-org/llama.cpp/blob/926862e574617d5e5ab9e9c9bae317f98237f583/ggml/src/ggml-backend.cpp"
],
"input_embedding": "explicit CUDA buffer override token_embd.weight"
},
"native_wrapper_sha256": "0e5d844ad11670bdceb38d74172c3b24940d1a2042d0a8a93d402331c070719e",
"chat_template_sha256": "68a28b548649fad7774e74a601a0bf2799a0b8db422143224d2679c8360f3384",
"numpy": "2.5.3",
"jinja2": "3.1.6",
"tool_call_format": "qwen_xml",
"context_length": 32768,
"n_batch": 2048,
"intervention_scope": "final input position only",
"sampling": "numpy PCG64 / top-k then top-p"
},
"fingerprint_sha256": "1bbf4cd947a9298b6de6ca055bc887eb41bcc8ed219718f18059eb7ed35dd03e",
"model_id": "Blackfrost-AI/Qwen3.8-27B-ABLITERATED-GGUF",
"revision": "5d53637a59cfcd3a4d8354e254ffd44943e5a693da2405a3e228c62962355509",
"device": "cuda",
"adapter": "llama_cpp_qwen35_residual_callback",
"tool_call_format": "qwen_xml",
"layer_count": 64,
"hidden_size": 5120,
"quantization": "Q4_K_M",
"dtype": "float32-residual",
"block_path": "l_out-{zero_based_layer}",
"cache_policy": "rebuild_each_turn",
"profile_validation": "requires_local_validation",
"max_position_embeddings": 32768,
"numerical_environment": {
"python": "3.13.7",
"platform": "Windows-11-10.0.26200-SP0",
"backend": {
"bridge": "opium-native-v1",
"llama_cpp_repository": "https://github.com/ggml-org/llama.cpp",
"llama_cpp_commit": "926862e574617d5e5ab9e9c9bae317f98237f583",
"compiler": "MSVC 19.43.34810",
"cuda": "13.0.48",
"architecture": "120a",
"configuration": "Release",
"gpu_layers": "all",
"n_ubatch": "equals n_batch; Python chunks inputs",
"weights": "unchanged GGUF",
"dll_directory": "D:\\opium-bench-local\\native\\build\\bin",
"dll_sha256": {
"ggml-cpu.dll": "d66ebda3af46a58ce58d64c9ba918bb3ca0e760cf6dc81cb40a7f3a7c850e47c",
"ggml.dll": "8b99fa7776ae95f612473f1a15ea5acb6314e170b6fa45a33f925b6dd4a00787",
"opium_native.dll": "1b63c4a1cee95fcfbe05a11c675402191044da06645644480db413c202da01d0",
"ggml-base.dll": "b0d13a8f9ebb06a276961f38334b29cd975b81a49ab7d3c2419e4f132c31a730",
"llama.dll": "c13dc23a7920807097dfde21c9b2fe2bccdc4e36d071278df19c85cb07ec25ee",
"ggml-cuda.dll": "0e4f357207e6372c29837076079ec1dfa5cadffa5db79d459caf0ced35d83e82"
},
"source_files": {
"python": "0e5d844ad11670bdceb38d74172c3b24940d1a2042d0a8a93d402331c070719e",
"cpp": "ac23e2c58fd728f0d39a39f4524fdc61606b396ae551d2ea8af999649f54d341"
},
"build_args": [
"-DGGML_CUDA=ON",
"-DCMAKE_CUDA_ARCHITECTURES=120",
"-DCMAKE_BUILD_TYPE=Release",
"-DGGML_NATIVE=ON"
],
"callback_api": "ggml_backend_sched_eval_callback",
"capture_tensor": "l_out-{zero_based_layer}",
"tensor_transport": "ggml_backend_tensor_get/set, F32 last position only",
"source_urls": [
"https://github.com/ggml-org/llama.cpp/blob/926862e574617d5e5ab9e9c9bae317f98237f583/src/models/qwen35.cpp",
"https://github.com/ggml-org/llama.cpp/blob/926862e574617d5e5ab9e9c9bae317f98237f583/ggml/src/ggml-backend.cpp"
],
"input_embedding": "explicit CUDA buffer override token_embd.weight"
},
"numpy": "2.5.3"
},
"source_sha256": {
"gguf_runtime.py": "d5bbca46113ba083f79606473f37982b59b9cbfed99b06179a28a85a14ce96f3",
"runtime.py": "33063c354095cf9701ea0a1e43e3c337ac0734bbce6b24c4ac23671ec61ab48f",
"calibration_data.py": "0b3e3d3933880ff1f1c0b65775e5024ffa178409f4559364b68f021b0338ffd7"
}
}