21 lines
15 KiB
JSON
21 lines
15 KiB
JSON
{"instance_id": "locomo_conv-50_q086", "cell": "no-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 2, "latency_ms": 166063, "prompt_tokens": 479, "completion_tokens": 1124, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F4\",\n \"rationale\": \"The model answered 'Surfing', an activity that does not appear in the ground-truth context, whereas the context explicitly indicates Calvin plans to try skiing after the tour by referencing a snowy peak and an image of a person on skis.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F4", "parsed_rationale": "The model answered 'Surfing', an activity that does not appear in the ground-truth context, whereas the context explicitly indicates Calvin plans to try skiing after the tour by referencing a snowy peak and an image of a person on skis."}
|
|
{"instance_id": "locomo_conv-26_q059", "cell": "no-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 50751, "prompt_tokens": 570, "completion_tokens": 1737, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F1\",\n \"rationale\": \"The model abstained by stating the answer cannot be determined, whereas the ground truth indicates Caroline is somewhat, but not extremely, religious.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F1", "parsed_rationale": "The model abstained by stating the answer cannot be determined, whereas the ground truth indicates Caroline is somewhat, but not extremely, religious."}
|
|
{"instance_id": "locomo_conv-44_q000", "cell": "no-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 25677, "prompt_tokens": 444, "completion_tokens": 698, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F4\",\n \"rationale\": \"The model's answer of 1954 is a date that does not appear in the ground-truth context and is factually inconsistent with the correct year 2020 implied by Audrey having had the dogs for 3 years as of March 2023.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F4", "parsed_rationale": "The model's answer of 1954 is a date that does not appear in the ground-truth context and is factually inconsistent with the correct year 2020 implied by Audrey having had the dogs for 3 years as of March 2023."}
|
|
{"instance_id": "locomo_conv-41_q080", "cell": "no-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 1, "latency_ms": 98768, "prompt_tokens": 495, "completion_tokens": 1031, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F4\",\n \"rationale\": \"The model claimed Maria made a wind chime, an entity not present in the context, instead of the ground-truth painting of a castle on a hill.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F4", "parsed_rationale": "The model claimed Maria made a wind chime, an entity not present in the context, instead of the ground-truth painting of a castle on a hill."}
|
|
{"instance_id": "locomo_conv-50_q086", "cell": "oracle-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 10807, "prompt_tokens": 480, "completion_tokens": 294, "raw_text": "{\n \"verdict\": \"correct\",\n \"failure_mode\": null,\n \"rationale\": \"The model's answer 'Skiing' exactly matches the ground-truth answer and is directly supported by the context mentioning a snowy peak and an image of a person on skis.\"\n}", "parsed_verdict": "correct", "parsed_failure_mode": null, "parsed_rationale": "The model's answer 'Skiing' exactly matches the ground-truth answer and is directly supported by the context mentioning a snowy peak and an image of a person on skis."}
|
|
{"instance_id": "locomo_conv-26_q059", "cell": "oracle-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 33265, "prompt_tokens": 566, "completion_tokens": 1486, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F2\",\n \"rationale\": \"The model correctly affirms Caroline is religious but omits the ground truth's nuance that she is only somewhat religious and not extremely religious.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F2", "parsed_rationale": "The model correctly affirms Caroline is religious but omits the ground truth's nuance that she is only somewhat religious and not extremely religious."}
|
|
{"instance_id": "locomo_conv-44_q000", "cell": "oracle-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 8245, "prompt_tokens": 444, "completion_tokens": 297, "raw_text": "{\n \"verdict\": \"correct\",\n \"failure_mode\": null,\n \"rationale\": \"The model correctly identified 2020 as the year Audrey adopted the dogs, matching the ground truth derived from the context.\"\n}", "parsed_verdict": "correct", "parsed_failure_mode": null, "parsed_rationale": "The model correctly identified 2020 as the year Audrey adopted the dogs, matching the ground truth derived from the context."}
|
|
{"instance_id": "locomo_conv-41_q080", "cell": "oracle-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 31274, "prompt_tokens": 493, "completion_tokens": 1013, "raw_text": "```json\n{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F2\",\n \"rationale\": \"The model correctly states that Maria made a painting, but fails to include the necessary detail that the painting depicts a castle on a hill.\"\n}\n```", "parsed_verdict": "incorrect", "parsed_failure_mode": "F2", "parsed_rationale": "The model correctly states that Maria made a painting, but fails to include the necessary detail that the painting depicts a castle on a hill."}
|
|
{"instance_id": "locomo_conv-50_q086", "cell": "full-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 9236, "prompt_tokens": 479, "completion_tokens": 264, "raw_text": "{\n \"verdict\": \"correct\",\n \"failure_mode\": null,\n \"rationale\": \"The model correctly identified 'skiing' as the sports activity Calvin plans to try after the tour, matching the ground-truth answer without adding incorrect claims.\"\n}", "parsed_verdict": "correct", "parsed_failure_mode": null, "parsed_rationale": "The model correctly identified 'skiing' as the sports activity Calvin plans to try after the tour, matching the ground-truth answer without adding incorrect claims."}
|
|
{"instance_id": "locomo_conv-26_q059", "cell": "full-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 25934, "prompt_tokens": 566, "completion_tokens": 833, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F1\",\n \"rationale\": \"The model explicitly answered 'unknown' rather than providing the supported nuanced assessment that Caroline is somewhat, but not extremely, religious.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F1", "parsed_rationale": "The model explicitly answered 'unknown' rather than providing the supported nuanced assessment that Caroline is somewhat, but not extremely, religious."}
|
|
{"instance_id": "locomo_conv-44_q000", "cell": "full-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 6796, "prompt_tokens": 444, "completion_tokens": 305, "raw_text": "{\n \"verdict\": \"correct\",\n \"failure_mode\": null,\n \"rationale\": \"The model correctly inferred from the context that Audrey adopted the dogs in 2020, matching the ground truth exactly.\"\n}", "parsed_verdict": "correct", "parsed_failure_mode": null, "parsed_rationale": "The model correctly inferred from the context that Audrey adopted the dogs in 2020, matching the ground truth exactly."}
|
|
{"instance_id": "locomo_conv-41_q080", "cell": "full-context", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 28923, "prompt_tokens": 493, "completion_tokens": 787, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F2\",\n \"rationale\": \"The model correctly states that Maria made a painting but omits the required detail that it depicts a castle on a hill.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F2", "parsed_rationale": "The model correctly states that Maria made a painting but omits the required detail that it depicts a castle on a hill."}
|
|
{"instance_id": "locomo_conv-50_q086", "cell": "retrieval", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 31953, "prompt_tokens": 479, "completion_tokens": 950, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F3\",\n \"rationale\": \"The model incorrectly claims the activity is not mentioned when the context explicitly references skiing through the image description of a person on skis and the mention of a snowy peak in connection with Calvin's post-tour plans.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F3", "parsed_rationale": "The model incorrectly claims the activity is not mentioned when the context explicitly references skiing through the image description of a person on skis and the mention of a snowy peak in connection with Calvin's post-tour plans."}
|
|
{"instance_id": "locomo_conv-26_q059", "cell": "retrieval", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 43785, "prompt_tokens": 566, "completion_tokens": 1286, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F3\",\n \"rationale\": \"The model incorrectly answered 'No,' asserting Caroline is not religious, whereas the ground truth indicates she is somewhat religious as shown by her creating a stained glass window for a local church.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F3", "parsed_rationale": "The model incorrectly answered 'No,' asserting Caroline is not religious, whereas the ground truth indicates she is somewhat religious as shown by her creating a stained glass window for a local church."}
|
|
{"instance_id": "locomo_conv-44_q000", "cell": "retrieval", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 1, "latency_ms": 110490, "prompt_tokens": 444, "completion_tokens": 1161, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F1\",\n \"rationale\": \"The model abstains by claiming the adoption year is not mentioned, but the context states Audrey has had the dogs for 3 years as of March 2023, which implies 2020.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F1", "parsed_rationale": "The model abstains by claiming the adoption year is not mentioned, but the context states Audrey has had the dogs for 3 years as of March 2023, which implies 2020."}
|
|
{"instance_id": "locomo_conv-41_q080", "cell": "retrieval", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 29546, "prompt_tokens": 493, "completion_tokens": 777, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F2\",\n \"rationale\": \"The model correctly states that Maria made a painting but omits the essential detail that it was a painting of a castle on a hill.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F2", "parsed_rationale": "The model correctly states that Maria made a painting but omits the essential detail that it was a painting of a castle on a hill."}
|
|
{"instance_id": "locomo_conv-50_q086", "cell": "agentic", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 57420, "prompt_tokens": 489, "completion_tokens": 1509, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F1\",\n \"rationale\": \"The model did not answer the question, instead returning a system message about reaching the maximum tool turns, effectively abstaining from providing the correct answer of skiing.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F1", "parsed_rationale": "The model did not answer the question, instead returning a system message about reaching the maximum tool turns, effectively abstaining from providing the correct answer of skiing."}
|
|
{"instance_id": "locomo_conv-26_q059", "cell": "agentic", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 28521, "prompt_tokens": 599, "completion_tokens": 1276, "raw_text": "```json\n{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F4\",\n \"rationale\": \"The model fabricated a claim about a Swedish necklace representing 'love, faith and strength' that does not appear in the provided ground-truth context.\"\n}\n```", "parsed_verdict": "incorrect", "parsed_failure_mode": "F4", "parsed_rationale": "The model fabricated a claim about a Swedish necklace representing 'love, faith and strength' that does not appear in the provided ground-truth context."}
|
|
{"instance_id": "locomo_conv-44_q000", "cell": "agentic", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 50434, "prompt_tokens": 454, "completion_tokens": 2083, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F1\",\n \"rationale\": \"The model did not provide the correct year and instead indicated it reached the maximum tool turn limit, effectively abstaining from answering the question.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F1", "parsed_rationale": "The model did not provide the correct year and instead indicated it reached the maximum tool turn limit, effectively abstaining from answering the question."}
|
|
{"instance_id": "locomo_conv-41_q080", "cell": "agentic", "provider": "kimi", "model_id": "kimi-k2.6", "routing": "direct", "http_status": 200, "error": null, "retries": 0, "latency_ms": 41180, "prompt_tokens": 493, "completion_tokens": 1121, "raw_text": "{\n \"verdict\": \"incorrect\",\n \"failure_mode\": \"F2\",\n \"rationale\": \"The model correctly states that Maria made a painting but omits the required detail that the painting depicted a castle on a hill.\"\n}", "parsed_verdict": "incorrect", "parsed_failure_mode": "F2", "parsed_rationale": "The model correctly states that Maria made a painting but omits the required detail that the painting depicted a castle on a hill."}
|