251 lines
10 KiB
JSON
251 lines
10 KiB
JSON
{
|
|
"bundle_sha256_without_self": "01b5de5925b7dcdee3df4a6dcdd25226963b77f6b4ae98e3f0573e4533cded93",
|
|
"created_at_utc": "2026-06-28T12:02:26.219+00:00",
|
|
"endpoint": "v1/chat/completions",
|
|
"request_payload": {
|
|
"frequency_penalty": 0.0,
|
|
"max_completion_tokens": 200,
|
|
"messages": [
|
|
{
|
|
"content": "You are a precise assistant. Answer in exactly three bullet points.",
|
|
"role": "system"
|
|
},
|
|
{
|
|
"content": "Explain why reproducibility matters in LLM experiments.",
|
|
"role": "user"
|
|
}
|
|
],
|
|
"model": "gpt-4.1-mini-2025-04-14",
|
|
"n": 1,
|
|
"presence_penalty": 0.0,
|
|
"seed": 42,
|
|
"stream": false,
|
|
"temperature": 0.0,
|
|
"top_p": 1.0
|
|
},
|
|
"request_payload_sha256": "bca98569dc74b6ce719cccef6bd96397691b1656b62bdd646c32e7408546c1b5",
|
|
"schema_version": "llm-repro-harness.v1",
|
|
"summary": {
|
|
"all_outputs_identical": false,
|
|
"all_response_models_identical": true,
|
|
"all_successful": true,
|
|
"all_system_fingerprints_identical": true,
|
|
"failed_trial_count": 0,
|
|
"successful_trial_count": 3,
|
|
"trial_count": 3,
|
|
"unique_content_sha256": [
|
|
"1599a4c919502157127047d0fe3362d9f804da7bb48e5ea144ae3cfd86f58dc9",
|
|
"33b675b86a0f7ff5206d75f0d81763c9928aec97614de5758100ed6bd1285db1",
|
|
"9f9d595b9bea32730625366fc01f1dc4d22854ad9cf583834106f86062d4e3d8"
|
|
],
|
|
"unique_response_models": [
|
|
"gpt-4.1-mini-2025-04-14"
|
|
],
|
|
"unique_system_fingerprints": [
|
|
"fp_8f3c0606ab"
|
|
]
|
|
},
|
|
"trials": [
|
|
{
|
|
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By providing clear methodologies and data, reproducibility enables others to build upon existing work, accelerating advancements in LLM development. \n- Enhances transparency: Reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases in LLM models.",
|
|
"content_sha256": "33b675b86a0f7ff5206d75f0d81763c9928aec97614de5758100ed6bd1285db1",
|
|
"elapsed_ms": 2005.2,
|
|
"finish_reason": "stop",
|
|
"ok": true,
|
|
"raw_response": {
|
|
"choices": [
|
|
{
|
|
"finish_reason": "stop",
|
|
"index": 0,
|
|
"logprobs": null,
|
|
"message": {
|
|
"annotations": [],
|
|
"audio": null,
|
|
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By providing clear methodologies and data, reproducibility enables others to build upon existing work, accelerating advancements in LLM development. \n- Enhances transparency: Reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases in LLM models.",
|
|
"function_call": null,
|
|
"refusal": null,
|
|
"role": "assistant",
|
|
"tool_calls": null
|
|
}
|
|
}
|
|
],
|
|
"created": 1782648142,
|
|
"id": "chatcmpl-DviVKgREgVmCI5v8UdAcIGdQxIo4G",
|
|
"model": "gpt-4.1-mini-2025-04-14",
|
|
"moderation": null,
|
|
"object": "chat.completion",
|
|
"service_tier": "default",
|
|
"system_fingerprint": "fp_8f3c0606ab",
|
|
"usage": {
|
|
"completion_tokens": 92,
|
|
"completion_tokens_details": {
|
|
"accepted_prediction_tokens": 0,
|
|
"audio_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"rejected_prediction_tokens": 0
|
|
},
|
|
"prompt_tokens": 34,
|
|
"prompt_tokens_details": {
|
|
"audio_tokens": 0,
|
|
"cached_tokens": 0
|
|
},
|
|
"total_tokens": 126
|
|
}
|
|
},
|
|
"response_id": "chatcmpl-DviVKgREgVmCI5v8UdAcIGdQxIo4G",
|
|
"response_model": "gpt-4.1-mini-2025-04-14",
|
|
"started_at_utc": "2026-06-28T12:02:21.404+00:00",
|
|
"system_fingerprint": "fp_8f3c0606ab",
|
|
"trial_index": 1,
|
|
"usage": {
|
|
"completion_tokens": 92,
|
|
"completion_tokens_details": {
|
|
"accepted_prediction_tokens": 0,
|
|
"audio_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"rejected_prediction_tokens": 0
|
|
},
|
|
"prompt_tokens": 34,
|
|
"prompt_tokens_details": {
|
|
"audio_tokens": 0,
|
|
"cached_tokens": 0
|
|
},
|
|
"total_tokens": 126
|
|
}
|
|
},
|
|
{
|
|
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By enabling others to build upon validated work, reproducibility accelerates innovation and improvement in LLM development. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases.",
|
|
"content_sha256": "9f9d595b9bea32730625366fc01f1dc4d22854ad9cf583834106f86062d4e3d8",
|
|
"elapsed_ms": 1477.74,
|
|
"finish_reason": "stop",
|
|
"ok": true,
|
|
"raw_response": {
|
|
"choices": [
|
|
{
|
|
"finish_reason": "stop",
|
|
"index": 0,
|
|
"logprobs": null,
|
|
"message": {
|
|
"annotations": [],
|
|
"audio": null,
|
|
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By enabling others to build upon validated work, reproducibility accelerates innovation and improvement in LLM development. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases.",
|
|
"function_call": null,
|
|
"refusal": null,
|
|
"role": "assistant",
|
|
"tool_calls": null
|
|
}
|
|
}
|
|
],
|
|
"created": 1782648143,
|
|
"id": "chatcmpl-DviVLp2D8EmQ3A6VCHfB8NCdCrp0q",
|
|
"model": "gpt-4.1-mini-2025-04-14",
|
|
"moderation": null,
|
|
"object": "chat.completion",
|
|
"service_tier": "default",
|
|
"system_fingerprint": "fp_8f3c0606ab",
|
|
"usage": {
|
|
"completion_tokens": 85,
|
|
"completion_tokens_details": {
|
|
"accepted_prediction_tokens": 0,
|
|
"audio_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"rejected_prediction_tokens": 0
|
|
},
|
|
"prompt_tokens": 34,
|
|
"prompt_tokens_details": {
|
|
"audio_tokens": 0,
|
|
"cached_tokens": 0
|
|
},
|
|
"total_tokens": 119
|
|
}
|
|
},
|
|
"response_id": "chatcmpl-DviVLp2D8EmQ3A6VCHfB8NCdCrp0q",
|
|
"response_model": "gpt-4.1-mini-2025-04-14",
|
|
"started_at_utc": "2026-06-28T12:02:23.410+00:00",
|
|
"system_fingerprint": "fp_8f3c0606ab",
|
|
"trial_index": 2,
|
|
"usage": {
|
|
"completion_tokens": 85,
|
|
"completion_tokens_details": {
|
|
"accepted_prediction_tokens": 0,
|
|
"audio_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"rejected_prediction_tokens": 0
|
|
},
|
|
"prompt_tokens": 34,
|
|
"prompt_tokens_details": {
|
|
"audio_tokens": 0,
|
|
"cached_tokens": 0
|
|
},
|
|
"total_tokens": 119
|
|
}
|
|
},
|
|
{
|
|
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: It enables the research community to build upon previous work confidently, accelerating advancements in LLM development and applications. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust, making it easier to identify limitations, biases, and areas for improvement in LLM models.",
|
|
"content_sha256": "1599a4c919502157127047d0fe3362d9f804da7bb48e5ea144ae3cfd86f58dc9",
|
|
"elapsed_ms": 1330.2,
|
|
"finish_reason": "stop",
|
|
"ok": true,
|
|
"raw_response": {
|
|
"choices": [
|
|
{
|
|
"finish_reason": "stop",
|
|
"index": 0,
|
|
"logprobs": null,
|
|
"message": {
|
|
"annotations": [],
|
|
"audio": null,
|
|
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: It enables the research community to build upon previous work confidently, accelerating advancements in LLM development and applications. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust, making it easier to identify limitations, biases, and areas for improvement in LLM models.",
|
|
"function_call": null,
|
|
"refusal": null,
|
|
"role": "assistant",
|
|
"tool_calls": null
|
|
}
|
|
}
|
|
],
|
|
"created": 1782648145,
|
|
"id": "chatcmpl-DviVN4bVJuFvdj82Jcu2uQN4xEcfi",
|
|
"model": "gpt-4.1-mini-2025-04-14",
|
|
"moderation": null,
|
|
"object": "chat.completion",
|
|
"service_tier": "default",
|
|
"system_fingerprint": "fp_8f3c0606ab",
|
|
"usage": {
|
|
"completion_tokens": 92,
|
|
"completion_tokens_details": {
|
|
"accepted_prediction_tokens": 0,
|
|
"audio_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"rejected_prediction_tokens": 0
|
|
},
|
|
"prompt_tokens": 34,
|
|
"prompt_tokens_details": {
|
|
"audio_tokens": 0,
|
|
"cached_tokens": 0
|
|
},
|
|
"total_tokens": 126
|
|
}
|
|
},
|
|
"response_id": "chatcmpl-DviVN4bVJuFvdj82Jcu2uQN4xEcfi",
|
|
"response_model": "gpt-4.1-mini-2025-04-14",
|
|
"started_at_utc": "2026-06-28T12:02:24.888+00:00",
|
|
"system_fingerprint": "fp_8f3c0606ab",
|
|
"trial_index": 3,
|
|
"usage": {
|
|
"completion_tokens": 92,
|
|
"completion_tokens_details": {
|
|
"accepted_prediction_tokens": 0,
|
|
"audio_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"rejected_prediction_tokens": 0
|
|
},
|
|
"prompt_tokens": 34,
|
|
"prompt_tokens_details": {
|
|
"audio_tokens": 0,
|
|
"cached_tokens": 0
|
|
},
|
|
"total_tokens": 126
|
|
}
|
|
}
|
|
]
|
|
} |