Files
llm-repro-ui/runs/run_2026-06-28T120226219+0000_bca98569dc74.json

251 lines
10 KiB
JSON

{
"bundle_sha256_without_self": "01b5de5925b7dcdee3df4a6dcdd25226963b77f6b4ae98e3f0573e4533cded93",
"created_at_utc": "2026-06-28T12:02:26.219+00:00",
"endpoint": "v1/chat/completions",
"request_payload": {
"frequency_penalty": 0.0,
"max_completion_tokens": 200,
"messages": [
{
"content": "You are a precise assistant. Answer in exactly three bullet points.",
"role": "system"
},
{
"content": "Explain why reproducibility matters in LLM experiments.",
"role": "user"
}
],
"model": "gpt-4.1-mini-2025-04-14",
"n": 1,
"presence_penalty": 0.0,
"seed": 42,
"stream": false,
"temperature": 0.0,
"top_p": 1.0
},
"request_payload_sha256": "bca98569dc74b6ce719cccef6bd96397691b1656b62bdd646c32e7408546c1b5",
"schema_version": "llm-repro-harness.v1",
"summary": {
"all_outputs_identical": false,
"all_response_models_identical": true,
"all_successful": true,
"all_system_fingerprints_identical": true,
"failed_trial_count": 0,
"successful_trial_count": 3,
"trial_count": 3,
"unique_content_sha256": [
"1599a4c919502157127047d0fe3362d9f804da7bb48e5ea144ae3cfd86f58dc9",
"33b675b86a0f7ff5206d75f0d81763c9928aec97614de5758100ed6bd1285db1",
"9f9d595b9bea32730625366fc01f1dc4d22854ad9cf583834106f86062d4e3d8"
],
"unique_response_models": [
"gpt-4.1-mini-2025-04-14"
],
"unique_system_fingerprints": [
"fp_8f3c0606ab"
]
},
"trials": [
{
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By providing clear methodologies and data, reproducibility enables others to build upon existing work, accelerating advancements in LLM development. \n- Enhances transparency: Reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases in LLM models.",
"content_sha256": "33b675b86a0f7ff5206d75f0d81763c9928aec97614de5758100ed6bd1285db1",
"elapsed_ms": 2005.2,
"finish_reason": "stop",
"ok": true,
"raw_response": {
"choices": [
{
"finish_reason": "stop",
"index": 0,
"logprobs": null,
"message": {
"annotations": [],
"audio": null,
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By providing clear methodologies and data, reproducibility enables others to build upon existing work, accelerating advancements in LLM development. \n- Enhances transparency: Reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases in LLM models.",
"function_call": null,
"refusal": null,
"role": "assistant",
"tool_calls": null
}
}
],
"created": 1782648142,
"id": "chatcmpl-DviVKgREgVmCI5v8UdAcIGdQxIo4G",
"model": "gpt-4.1-mini-2025-04-14",
"moderation": null,
"object": "chat.completion",
"service_tier": "default",
"system_fingerprint": "fp_8f3c0606ab",
"usage": {
"completion_tokens": 92,
"completion_tokens_details": {
"accepted_prediction_tokens": 0,
"audio_tokens": 0,
"reasoning_tokens": 0,
"rejected_prediction_tokens": 0
},
"prompt_tokens": 34,
"prompt_tokens_details": {
"audio_tokens": 0,
"cached_tokens": 0
},
"total_tokens": 126
}
},
"response_id": "chatcmpl-DviVKgREgVmCI5v8UdAcIGdQxIo4G",
"response_model": "gpt-4.1-mini-2025-04-14",
"started_at_utc": "2026-06-28T12:02:21.404+00:00",
"system_fingerprint": "fp_8f3c0606ab",
"trial_index": 1,
"usage": {
"completion_tokens": 92,
"completion_tokens_details": {
"accepted_prediction_tokens": 0,
"audio_tokens": 0,
"reasoning_tokens": 0,
"rejected_prediction_tokens": 0
},
"prompt_tokens": 34,
"prompt_tokens_details": {
"audio_tokens": 0,
"cached_tokens": 0
},
"total_tokens": 126
}
},
{
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By enabling others to build upon validated work, reproducibility accelerates innovation and improvement in LLM development. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases.",
"content_sha256": "9f9d595b9bea32730625366fc01f1dc4d22854ad9cf583834106f86062d4e3d8",
"elapsed_ms": 1477.74,
"finish_reason": "stop",
"ok": true,
"raw_response": {
"choices": [
{
"finish_reason": "stop",
"index": 0,
"logprobs": null,
"message": {
"annotations": [],
"audio": null,
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By enabling others to build upon validated work, reproducibility accelerates innovation and improvement in LLM development. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases.",
"function_call": null,
"refusal": null,
"role": "assistant",
"tool_calls": null
}
}
],
"created": 1782648143,
"id": "chatcmpl-DviVLp2D8EmQ3A6VCHfB8NCdCrp0q",
"model": "gpt-4.1-mini-2025-04-14",
"moderation": null,
"object": "chat.completion",
"service_tier": "default",
"system_fingerprint": "fp_8f3c0606ab",
"usage": {
"completion_tokens": 85,
"completion_tokens_details": {
"accepted_prediction_tokens": 0,
"audio_tokens": 0,
"reasoning_tokens": 0,
"rejected_prediction_tokens": 0
},
"prompt_tokens": 34,
"prompt_tokens_details": {
"audio_tokens": 0,
"cached_tokens": 0
},
"total_tokens": 119
}
},
"response_id": "chatcmpl-DviVLp2D8EmQ3A6VCHfB8NCdCrp0q",
"response_model": "gpt-4.1-mini-2025-04-14",
"started_at_utc": "2026-06-28T12:02:23.410+00:00",
"system_fingerprint": "fp_8f3c0606ab",
"trial_index": 2,
"usage": {
"completion_tokens": 85,
"completion_tokens_details": {
"accepted_prediction_tokens": 0,
"audio_tokens": 0,
"reasoning_tokens": 0,
"rejected_prediction_tokens": 0
},
"prompt_tokens": 34,
"prompt_tokens_details": {
"audio_tokens": 0,
"cached_tokens": 0
},
"total_tokens": 119
}
},
{
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: It enables the research community to build upon previous work confidently, accelerating advancements in LLM development and applications. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust, making it easier to identify limitations, biases, and areas for improvement in LLM models.",
"content_sha256": "1599a4c919502157127047d0fe3362d9f804da7bb48e5ea144ae3cfd86f58dc9",
"elapsed_ms": 1330.2,
"finish_reason": "stop",
"ok": true,
"raw_response": {
"choices": [
{
"finish_reason": "stop",
"index": 0,
"logprobs": null,
"message": {
"annotations": [],
"audio": null,
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: It enables the research community to build upon previous work confidently, accelerating advancements in LLM development and applications. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust, making it easier to identify limitations, biases, and areas for improvement in LLM models.",
"function_call": null,
"refusal": null,
"role": "assistant",
"tool_calls": null
}
}
],
"created": 1782648145,
"id": "chatcmpl-DviVN4bVJuFvdj82Jcu2uQN4xEcfi",
"model": "gpt-4.1-mini-2025-04-14",
"moderation": null,
"object": "chat.completion",
"service_tier": "default",
"system_fingerprint": "fp_8f3c0606ab",
"usage": {
"completion_tokens": 92,
"completion_tokens_details": {
"accepted_prediction_tokens": 0,
"audio_tokens": 0,
"reasoning_tokens": 0,
"rejected_prediction_tokens": 0
},
"prompt_tokens": 34,
"prompt_tokens_details": {
"audio_tokens": 0,
"cached_tokens": 0
},
"total_tokens": 126
}
},
"response_id": "chatcmpl-DviVN4bVJuFvdj82Jcu2uQN4xEcfi",
"response_model": "gpt-4.1-mini-2025-04-14",
"started_at_utc": "2026-06-28T12:02:24.888+00:00",
"system_fingerprint": "fp_8f3c0606ab",
"trial_index": 3,
"usage": {
"completion_tokens": 92,
"completion_tokens_details": {
"accepted_prediction_tokens": 0,
"audio_tokens": 0,
"reasoning_tokens": 0,
"rejected_prediction_tokens": 0
},
"prompt_tokens": 34,
"prompt_tokens_details": {
"audio_tokens": 0,
"cached_tokens": 0
},
"total_tokens": 126
}
}
]
}