8 lines
5.4 KiB
JSON
8 lines
5.4 KiB
JSON
{"instance_id": "locomo_conv-43_q155", "cell": "agentic", "provider": "deepseek", "model_id": "deepseek-v4-pro", "routing": "direct", "max_tokens": 2048, "http_status": 200, "error": null, "retries": 0, "latency_ms": 42713, "prompt_tokens": 457, "completion_tokens": 2000, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F3\",\n \"rationale\": \"The model answered 'great', but the ground truth answer is 'It was a huge success.'; while 'great' appears in the context, it does not contain the required fact.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F3", "parsed_rationale": "The model answered 'great', but the ground truth answer is 'It was a huge success.'; while 'great' appears in the context, it does not contain the required fact.", "opus_verdict_ref": "correct", "gpt_verdict_ref": "incorrect"}
|
|
{"instance_id": "locomo_conv-50_q015", "cell": "agentic", "provider": "deepseek", "model_id": "deepseek-v4-pro", "routing": "direct", "max_tokens": 2048, "http_status": 200, "error": null, "retries": 0, "latency_ms": 30086, "prompt_tokens": 556, "completion_tokens": 1053, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F4\",\n \"rationale\": \"The model adds 'recharge' which is not supported by the ground truth or context, making it a fabricated claim, while missing the required fact 'calms'.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F4", "parsed_rationale": "The model adds 'recharge' which is not supported by the ground truth or context, making it a fabricated claim, while missing the required fact 'calms'.", "opus_verdict_ref": "correct", "gpt_verdict_ref": "incorrect"}
|
|
{"instance_id": "locomo_conv-43_q155", "cell": "full-context", "provider": "deepseek", "model_id": "deepseek-v4-pro", "routing": "direct", "max_tokens": 2048, "http_status": 200, "error": null, "retries": 0, "latency_ms": 7693, "prompt_tokens": 457, "completion_tokens": 246, "raw_text": "{\n \"verdict\": \"correct\",\n \"failure_mode\": null,\n \"rationale\": \"The model's answer 'great' is a direct synonym for John's expressed feeling 'It was great!', which accurately answers the question about how he felt.\"\n}", "parsed_verdict": "correct", "parsed_failure_mode": null, "parsed_rationale": "The model's answer 'great' is a direct synonym for John's expressed feeling 'It was great!', which accurately answers the question about how he felt.", "opus_verdict_ref": "correct", "gpt_verdict_ref": "incorrect"}
|
|
{"instance_id": "locomo_conv-50_q015", "cell": "full-context", "provider": "deepseek", "model_id": "deepseek-v4-pro", "routing": "direct", "max_tokens": 2048, "http_status": 200, "error": null, "retries": 0, "latency_ms": 15014, "prompt_tokens": 554, "completion_tokens": 522, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F2\",\n \"rationale\": \"The model's answer 'to relax' captures only part of the ground truth reason; it omits that visiting parks also calms him.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F2", "parsed_rationale": "The model's answer 'to relax' captures only part of the ground truth reason; it omits that visiting parks also calms him.", "opus_verdict_ref": "correct", "gpt_verdict_ref": "incorrect"}
|
|
{"instance_id": "locomo_conv-43_q155", "cell": "oracle-context", "provider": "deepseek", "model_id": "deepseek-v4-pro", "routing": "direct", "max_tokens": 2048, "http_status": 200, "error": null, "retries": 0, "latency_ms": 24898, "prompt_tokens": 457, "completion_tokens": 886, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F2\",\n \"rationale\": \"The model's answer 'Great' captures only part of the ground-truth answer, omitting the 'huge success' aspect.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F2", "parsed_rationale": "The model's answer 'Great' captures only part of the ground-truth answer, omitting the 'huge success' aspect.", "opus_verdict_ref": "correct", "gpt_verdict_ref": "incorrect"}
|
|
{"instance_id": "locomo_conv-47_q092", "cell": "oracle-context", "provider": "deepseek", "model_id": "deepseek-v4-pro", "routing": "direct", "max_tokens": 2048, "http_status": 200, "error": null, "retries": 0, "latency_ms": 13424, "prompt_tokens": 460, "completion_tokens": 608, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F2\",\n \"rationale\": \"The model's answer 'An application' omits the required detail that it was a computer application on smartphones, making it incomplete.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F2", "parsed_rationale": "The model's answer 'An application' omits the required detail that it was a computer application on smartphones, making it incomplete.", "opus_verdict_ref": "correct", "gpt_verdict_ref": "incorrect"}
|
|
{"instance_id": "locomo_conv-50_q015", "cell": "oracle-context", "provider": "deepseek", "model_id": "deepseek-v4-pro", "routing": "direct", "max_tokens": 2048, "http_status": 200, "error": null, "retries": 0, "latency_ms": 9419, "prompt_tokens": 554, "completion_tokens": 398, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F2\",\n \"rationale\": \"The model's answer captures relaxation but omits the calming aspect, which is part of the ground truth reason.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F2", "parsed_rationale": "The model's answer captures relaxation but omits the calming aspect, which is part of the ground truth reason.", "opus_verdict_ref": "correct", "gpt_verdict_ref": "incorrect"}
|