Initial commit, very early alpha
This commit is contained in:
251
runs/run_2026-06-28T120226219+0000_bca98569dc74.json
Normal file
251
runs/run_2026-06-28T120226219+0000_bca98569dc74.json
Normal file
@@ -0,0 +1,251 @@
|
||||
{
|
||||
"bundle_sha256_without_self": "01b5de5925b7dcdee3df4a6dcdd25226963b77f6b4ae98e3f0573e4533cded93",
|
||||
"created_at_utc": "2026-06-28T12:02:26.219+00:00",
|
||||
"endpoint": "v1/chat/completions",
|
||||
"request_payload": {
|
||||
"frequency_penalty": 0.0,
|
||||
"max_completion_tokens": 200,
|
||||
"messages": [
|
||||
{
|
||||
"content": "You are a precise assistant. Answer in exactly three bullet points.",
|
||||
"role": "system"
|
||||
},
|
||||
{
|
||||
"content": "Explain why reproducibility matters in LLM experiments.",
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"model": "gpt-4.1-mini-2025-04-14",
|
||||
"n": 1,
|
||||
"presence_penalty": 0.0,
|
||||
"seed": 42,
|
||||
"stream": false,
|
||||
"temperature": 0.0,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"request_payload_sha256": "bca98569dc74b6ce719cccef6bd96397691b1656b62bdd646c32e7408546c1b5",
|
||||
"schema_version": "llm-repro-harness.v1",
|
||||
"summary": {
|
||||
"all_outputs_identical": false,
|
||||
"all_response_models_identical": true,
|
||||
"all_successful": true,
|
||||
"all_system_fingerprints_identical": true,
|
||||
"failed_trial_count": 0,
|
||||
"successful_trial_count": 3,
|
||||
"trial_count": 3,
|
||||
"unique_content_sha256": [
|
||||
"1599a4c919502157127047d0fe3362d9f804da7bb48e5ea144ae3cfd86f58dc9",
|
||||
"33b675b86a0f7ff5206d75f0d81763c9928aec97614de5758100ed6bd1285db1",
|
||||
"9f9d595b9bea32730625366fc01f1dc4d22854ad9cf583834106f86062d4e3d8"
|
||||
],
|
||||
"unique_response_models": [
|
||||
"gpt-4.1-mini-2025-04-14"
|
||||
],
|
||||
"unique_system_fingerprints": [
|
||||
"fp_8f3c0606ab"
|
||||
]
|
||||
},
|
||||
"trials": [
|
||||
{
|
||||
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By providing clear methodologies and data, reproducibility enables others to build upon existing work, accelerating advancements in LLM development. \n- Enhances transparency: Reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases in LLM models.",
|
||||
"content_sha256": "33b675b86a0f7ff5206d75f0d81763c9928aec97614de5758100ed6bd1285db1",
|
||||
"elapsed_ms": 2005.2,
|
||||
"finish_reason": "stop",
|
||||
"ok": true,
|
||||
"raw_response": {
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "stop",
|
||||
"index": 0,
|
||||
"logprobs": null,
|
||||
"message": {
|
||||
"annotations": [],
|
||||
"audio": null,
|
||||
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By providing clear methodologies and data, reproducibility enables others to build upon existing work, accelerating advancements in LLM development. \n- Enhances transparency: Reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases in LLM models.",
|
||||
"function_call": null,
|
||||
"refusal": null,
|
||||
"role": "assistant",
|
||||
"tool_calls": null
|
||||
}
|
||||
}
|
||||
],
|
||||
"created": 1782648142,
|
||||
"id": "chatcmpl-DviVKgREgVmCI5v8UdAcIGdQxIo4G",
|
||||
"model": "gpt-4.1-mini-2025-04-14",
|
||||
"moderation": null,
|
||||
"object": "chat.completion",
|
||||
"service_tier": "default",
|
||||
"system_fingerprint": "fp_8f3c0606ab",
|
||||
"usage": {
|
||||
"completion_tokens": 92,
|
||||
"completion_tokens_details": {
|
||||
"accepted_prediction_tokens": 0,
|
||||
"audio_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"rejected_prediction_tokens": 0
|
||||
},
|
||||
"prompt_tokens": 34,
|
||||
"prompt_tokens_details": {
|
||||
"audio_tokens": 0,
|
||||
"cached_tokens": 0
|
||||
},
|
||||
"total_tokens": 126
|
||||
}
|
||||
},
|
||||
"response_id": "chatcmpl-DviVKgREgVmCI5v8UdAcIGdQxIo4G",
|
||||
"response_model": "gpt-4.1-mini-2025-04-14",
|
||||
"started_at_utc": "2026-06-28T12:02:21.404+00:00",
|
||||
"system_fingerprint": "fp_8f3c0606ab",
|
||||
"trial_index": 1,
|
||||
"usage": {
|
||||
"completion_tokens": 92,
|
||||
"completion_tokens_details": {
|
||||
"accepted_prediction_tokens": 0,
|
||||
"audio_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"rejected_prediction_tokens": 0
|
||||
},
|
||||
"prompt_tokens": 34,
|
||||
"prompt_tokens_details": {
|
||||
"audio_tokens": 0,
|
||||
"cached_tokens": 0
|
||||
},
|
||||
"total_tokens": 126
|
||||
}
|
||||
},
|
||||
{
|
||||
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By enabling others to build upon validated work, reproducibility accelerates innovation and improvement in LLM development. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases.",
|
||||
"content_sha256": "9f9d595b9bea32730625366fc01f1dc4d22854ad9cf583834106f86062d4e3d8",
|
||||
"elapsed_ms": 1477.74,
|
||||
"finish_reason": "stop",
|
||||
"ok": true,
|
||||
"raw_response": {
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "stop",
|
||||
"index": 0,
|
||||
"logprobs": null,
|
||||
"message": {
|
||||
"annotations": [],
|
||||
"audio": null,
|
||||
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: By enabling others to build upon validated work, reproducibility accelerates innovation and improvement in LLM development. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust in the research community, helping to identify limitations and biases.",
|
||||
"function_call": null,
|
||||
"refusal": null,
|
||||
"role": "assistant",
|
||||
"tool_calls": null
|
||||
}
|
||||
}
|
||||
],
|
||||
"created": 1782648143,
|
||||
"id": "chatcmpl-DviVLp2D8EmQ3A6VCHfB8NCdCrp0q",
|
||||
"model": "gpt-4.1-mini-2025-04-14",
|
||||
"moderation": null,
|
||||
"object": "chat.completion",
|
||||
"service_tier": "default",
|
||||
"system_fingerprint": "fp_8f3c0606ab",
|
||||
"usage": {
|
||||
"completion_tokens": 85,
|
||||
"completion_tokens_details": {
|
||||
"accepted_prediction_tokens": 0,
|
||||
"audio_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"rejected_prediction_tokens": 0
|
||||
},
|
||||
"prompt_tokens": 34,
|
||||
"prompt_tokens_details": {
|
||||
"audio_tokens": 0,
|
||||
"cached_tokens": 0
|
||||
},
|
||||
"total_tokens": 119
|
||||
}
|
||||
},
|
||||
"response_id": "chatcmpl-DviVLp2D8EmQ3A6VCHfB8NCdCrp0q",
|
||||
"response_model": "gpt-4.1-mini-2025-04-14",
|
||||
"started_at_utc": "2026-06-28T12:02:23.410+00:00",
|
||||
"system_fingerprint": "fp_8f3c0606ab",
|
||||
"trial_index": 2,
|
||||
"usage": {
|
||||
"completion_tokens": 85,
|
||||
"completion_tokens_details": {
|
||||
"accepted_prediction_tokens": 0,
|
||||
"audio_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"rejected_prediction_tokens": 0
|
||||
},
|
||||
"prompt_tokens": 34,
|
||||
"prompt_tokens_details": {
|
||||
"audio_tokens": 0,
|
||||
"cached_tokens": 0
|
||||
},
|
||||
"total_tokens": 119
|
||||
}
|
||||
},
|
||||
{
|
||||
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: It enables the research community to build upon previous work confidently, accelerating advancements in LLM development and applications. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust, making it easier to identify limitations, biases, and areas for improvement in LLM models.",
|
||||
"content_sha256": "1599a4c919502157127047d0fe3362d9f804da7bb48e5ea144ae3cfd86f58dc9",
|
||||
"elapsed_ms": 1330.2,
|
||||
"finish_reason": "stop",
|
||||
"ok": true,
|
||||
"raw_response": {
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "stop",
|
||||
"index": 0,
|
||||
"logprobs": null,
|
||||
"message": {
|
||||
"annotations": [],
|
||||
"audio": null,
|
||||
"content": "- Ensures reliability: Reproducibility allows researchers to verify results by independently replicating experiments, confirming that findings are not due to chance or errors. \n- Facilitates progress: It enables the research community to build upon previous work confidently, accelerating advancements in LLM development and applications. \n- Enhances transparency: Clear, reproducible experiments promote openness and trust, making it easier to identify limitations, biases, and areas for improvement in LLM models.",
|
||||
"function_call": null,
|
||||
"refusal": null,
|
||||
"role": "assistant",
|
||||
"tool_calls": null
|
||||
}
|
||||
}
|
||||
],
|
||||
"created": 1782648145,
|
||||
"id": "chatcmpl-DviVN4bVJuFvdj82Jcu2uQN4xEcfi",
|
||||
"model": "gpt-4.1-mini-2025-04-14",
|
||||
"moderation": null,
|
||||
"object": "chat.completion",
|
||||
"service_tier": "default",
|
||||
"system_fingerprint": "fp_8f3c0606ab",
|
||||
"usage": {
|
||||
"completion_tokens": 92,
|
||||
"completion_tokens_details": {
|
||||
"accepted_prediction_tokens": 0,
|
||||
"audio_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"rejected_prediction_tokens": 0
|
||||
},
|
||||
"prompt_tokens": 34,
|
||||
"prompt_tokens_details": {
|
||||
"audio_tokens": 0,
|
||||
"cached_tokens": 0
|
||||
},
|
||||
"total_tokens": 126
|
||||
}
|
||||
},
|
||||
"response_id": "chatcmpl-DviVN4bVJuFvdj82Jcu2uQN4xEcfi",
|
||||
"response_model": "gpt-4.1-mini-2025-04-14",
|
||||
"started_at_utc": "2026-06-28T12:02:24.888+00:00",
|
||||
"system_fingerprint": "fp_8f3c0606ab",
|
||||
"trial_index": 3,
|
||||
"usage": {
|
||||
"completion_tokens": 92,
|
||||
"completion_tokens_details": {
|
||||
"accepted_prediction_tokens": 0,
|
||||
"audio_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"rejected_prediction_tokens": 0
|
||||
},
|
||||
"prompt_tokens": 34,
|
||||
"prompt_tokens_details": {
|
||||
"audio_tokens": 0,
|
||||
"cached_tokens": 0
|
||||
},
|
||||
"total_tokens": 126
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
Reference in New Issue
Block a user