{"id":48947,"topic":"ai","source":"STAT","title":"Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated - STAT","url":"https://www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","url_hash":"bd64efb7c6a68896165cafc8fa1a24c5bf3dbaab","author":"","summary":"<a href=\"https://news.google.com/rss/articles/CBMipAFBVV95cUxNOWNiT2R6ZDF4S2QyWmZGbEhvOFIwbllXR09qZzR0SEl3ZGJOb2xhT2EtdGo3TEpNVElRZXUzbElWYjlpZFpzUi1ja3FDZG53N2JPR2NrcFgxYy1MeGpfNXg2dzhjbDdOcTdIZERTUHlFRnpSQ0RkVFl1TEo0Z2Roa1otalBaNm5PRllRZ1dWc1hQRVh2N1pvUHRQUWs3LVg1eGdqVQ?oc=5\" target=\"_blank\">Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated</a>&nbsp;&nbsp;<font color=\"#6f6f6f\">STAT</font>","content":null,"image_url":null,"lang":"en","published_at":"2026-07-29T13:33:25+00:00","fetched_at":"2026-07-29T14:15:03+00:00","status":"read","starred":0,"extract_state":"failed","summary_auto":null,"cluster_id":null,"extract_retries":2,"extract_error":"http_403","contract_version":"news_item.v1","format_contract_version":"news_item_formats.v1","dedup_url":"https://www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","quality_profile":{"profile_version":"extraction_quality.v2","bucket":"failed","confidence":0.0,"failure_kind":"http","retryable":false,"retry_after_attempts":0,"reason":"Low confidence: HTTP extraction failure (http_403).","operator_guidance":{"severity":"action_required","recommended_action":"manual_review","next_step":"Open diagnostics and review the original source manually before use.","operator_label":"Manual review","can_retry":false,"can_use_summary":false,"diagnostics_required":true},"content_depth":{"contract_version":"content_depth.v1","category":"summary_only","label":"Summary only","has_full_text":false,"has_summary":true,"content_length":0,"summary_length":417,"usable_text_length":417,"source_field":"summary"},"legacy_collapsed":false,"signals":{"extract_state":"failed","extract_error":"http_403","extract_retries":2,"content_length":0,"summary_length":417}},"news_item":{"id":48947,"canonical_url":"https://www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","source_url":"https://www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","title":"Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated - STAT","source_name":"STAT","author":null,"published_at":"2026-07-29T13:33:25+00:00","locale":"en","topic":"ai","tags":[],"rss_summary":"<a href=\"https://news.google.com/rss/articles/CBMipAFBVV95cUxNOWNiT2R6ZDF4S2QyWmZGbEhvOFIwbllXR09qZzR0SEl3ZGJOb2xhT2EtdGo3TEpNVElRZXUzbElWYjlpZFpzUi1ja3FDZG53N2JPR2NrcFgxYy1MeGpfNXg2dzhjbDdOcTdIZERTUHlFRnpSQ0RkVFl1TEo0Z2Roa1otalBaNm5PRllRZ1dWc1hQRVh2N1pvUHRQUWs3LVg1eGdqVQ?oc=5\" target=\"_blank\">Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated</a>&nbsp;&nbsp;<font color=\"#6f6f6f\">STAT</font>","full_text":null,"excerpt":"<a href=\"https://news.google.com/rss/articles/CBMipAFBVV95cUxNOWNiT2R6ZDF4S2QyWmZGbEhvOFIwbllXR09qZzR0SEl3ZGJOb2xhT2EtdGo3TEpNVElRZXUzbElWYjlpZFpzUi1ja3FDZG53N2JPR2NrcFgxYy1MeGpfNXg2dzhjbDdOcTdIZERTUHlFRnpSQ0RkVFl1TEo0Z2Roa1otalBaNm5PRllRZ1dWc1hQRVh2N1pvUHRQUWs3LVg1eGdqVQ?oc=5\" target=\"_blank\">Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated</a>&nbsp;&nbsp;<font color=\"#6f6f6f\">STAT</font>","extraction":{"state":"failed","confidence":0.0,"error":"http_403","explanation":"Low confidence: HTTP extraction failure (http_403).","diagnostics_url":"/api/diagnose?url=https%3A//www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","quality_profile":{"profile_version":"extraction_quality.v2","bucket":"failed","confidence":0.0,"failure_kind":"http","retryable":false,"retry_after_attempts":0,"reason":"Low confidence: HTTP extraction failure (http_403).","operator_guidance":{"severity":"action_required","recommended_action":"manual_review","next_step":"Open diagnostics and review the original source manually before use.","operator_label":"Manual review","can_retry":false,"can_use_summary":false,"diagnostics_required":true},"content_depth":{"contract_version":"content_depth.v1","category":"summary_only","label":"Summary only","has_full_text":false,"has_summary":true,"content_length":0,"summary_length":417,"usable_text_length":417,"source_field":"summary"},"legacy_collapsed":false,"signals":{"extract_state":"failed","extract_error":"http_403","extract_retries":2,"content_length":0,"summary_length":417}}},"display_formats":["compact","card","full","digest_section","json"]},"daily_stack_record":{"title":"Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated - STAT","url":"https://www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","summary":"<a href=\"https://news.google.com/rss/articles/CBMipAFBVV95cUxNOWNiT2R6ZDF4S2QyWmZGbEhvOFIwbllXR09qZzR0SEl3ZGJOb2xhT2EtdGo3TEpNVElRZXUzbElWYjlpZFpzUi1ja3FDZG53N2JPR2NrcFgxYy1MeGpfNXg2dzhjbDdOcTdIZERTUHlFRnpSQ0RkVFl1TEo0Z2Roa1otalBaNm5PRllRZ1dWc1hQRVh2N1pvUHRQUWs3LVg1eGdqVQ?oc=5\" target=\"_blank\">Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated</a>&nbsp;&nbsp;<font color=\"#6f6f6f\">STAT</font>","source":"STAT","date":"2026-07-29T13:33:25+00:00","content":"","confidence":0.0,"diagnostics_url":"/api/diagnose?url=https%3A//www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","quality_bucket":"failed","failure_kind":"http","retryable":false,"quality_reason":"Low confidence: HTTP extraction failure (http_403).","quality_profile":{"profile_version":"extraction_quality.v2","bucket":"failed","confidence":0.0,"failure_kind":"http","retryable":false,"retry_after_attempts":0,"reason":"Low confidence: HTTP extraction failure (http_403).","operator_guidance":{"severity":"action_required","recommended_action":"manual_review","next_step":"Open diagnostics and review the original source manually before use.","operator_label":"Manual review","can_retry":false,"can_use_summary":false,"diagnostics_required":true},"content_depth":{"contract_version":"content_depth.v1","category":"summary_only","label":"Summary only","has_full_text":false,"has_summary":true,"content_length":0,"summary_length":417,"usable_text_length":417,"source_field":"summary"},"legacy_collapsed":false,"signals":{"extract_state":"failed","extract_error":"http_403","extract_retries":2,"content_length":0,"summary_length":417}},"tags":[]},"fallback_formats":["markdown","json","html"],"actions":{"read":"/item/48947","export_markdown":"/api/items/48947/export?format=markdown","export_json":"/api/items/48947/export?format=json","diagnose":"/api/diagnose?url=https%3A//www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/"},"formats":{"full":{"id":48947,"title":"Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated - STAT","url":"https://www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","source":"STAT","author":null,"published_at":"2026-07-29T13:33:25+00:00","locale":"en","topic":"ai","tags":[],"excerpt":"<a href=\"https://news.google.com/rss/articles/CBMipAFBVV95cUxNOWNiT2R6ZDF4S2QyWmZGbEhvOFIwbllXR09qZzR0SEl3ZGJOb2xhT2EtdGo3TEpNVElRZXUzbElWYjlpZFpzUi1ja3FDZG53N2JPR2NrcFgxYy1MeGpfNXg2dzhjbDdOcTdIZERTUHlFRnpSQ0RkVFl1TEo0Z2Roa1otalBaNm5PRllRZ1dWc1hQRVh2N1pvUHRQUWs3LVg1eGdqVQ?oc=5\" target=\"_blank\">Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated</a>&nbsp;&nbsp;<font color=\"#6f6f6f\">STAT</font>","full_text":null,"reading_time_min":1,"extraction":{"state":"failed","confidence":0.0,"error":"http_403","explanation":"Low confidence: HTTP extraction failure (http_403).","diagnostics_url":"/api/diagnose?url=https%3A//www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","quality_profile":{"profile_version":"extraction_quality.v2","bucket":"failed","confidence":0.0,"failure_kind":"http","retryable":false,"retry_after_attempts":0,"reason":"Low confidence: HTTP extraction failure (http_403).","operator_guidance":{"severity":"action_required","recommended_action":"manual_review","next_step":"Open diagnostics and review the original source manually before use.","operator_label":"Manual review","can_retry":false,"can_use_summary":false,"diagnostics_required":true},"content_depth":{"contract_version":"content_depth.v1","category":"summary_only","label":"Summary only","has_full_text":false,"has_summary":true,"content_length":0,"summary_length":417,"usable_text_length":417,"source_field":"summary"},"legacy_collapsed":false,"signals":{"extract_state":"failed","extract_error":"http_403","extract_retries":2,"content_length":0,"summary_length":417}}},"quality_profile":{"profile_version":"extraction_quality.v2","bucket":"failed","confidence":0.0,"failure_kind":"http","retryable":false,"retry_after_attempts":0,"reason":"Low confidence: HTTP extraction failure (http_403).","operator_guidance":{"severity":"action_required","recommended_action":"manual_review","next_step":"Open diagnostics and review the original source manually before use.","operator_label":"Manual review","can_retry":false,"can_use_summary":false,"diagnostics_required":true},"content_depth":{"contract_version":"content_depth.v1","category":"summary_only","label":"Summary only","has_full_text":false,"has_summary":true,"content_length":0,"summary_length":417,"usable_text_length":417,"source_field":"summary"},"legacy_collapsed":false,"signals":{"extract_state":"failed","extract_error":"http_403","extract_retries":2,"content_length":0,"summary_length":417}},"actions":{"read":"/item/48947","export_markdown":"/api/items/48947/export?format=markdown","export_json":"/api/items/48947/export?format=json","diagnose":"/api/diagnose?url=https%3A//www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/"}},"digest":{"id":48947,"title":"Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated - STAT","url":"https://www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","source":"STAT","topic":"ai","published_at":"2026-07-29T13:33:25+00:00","excerpt":"<a href=\"https://news.google.com/rss/articles/CBMipAFBVV95cUxNOWNiT2R6ZDF4S2QyWmZGbEhvOFIwbllXR09qZzR0SEl3ZGJOb2xhT2EtdGo3TEpNVElRZXUzbElWYjlpZFpzUi1ja3FDZG53N2JPR2NrcFgxYy1MeGpfNXg2dzhjbDdOcTdIZERTUHlFRnpSQ0RkVFl1TEo0Z2Roa1otalBaNm5PRllRZ1dWc1hQRVh2N1pvUHRQUWs3LVg1eGdqVQ?oc=5\"…","quality_bucket":"failed","quality_reason":"Low confidence: HTTP extraction failure (http_403).","reading_time_min":1,"cluster_id":null},"card":{"display_title":"Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated - STAT","subtitle":"STAT · 2026-07-29","summary":"<a…","badges":["quality:failed","issue:http"],"links":{"read":"/item/48947","original":"https://www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","diagnose":"/api/diagnose?url=https%3A//www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/"},"quality_warning":"Low confidence: HTTP extraction failure (http_403)."},"export":{"title":"Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated - STAT","url":"https://www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","summary":"<a href=\"https://news.google.com/rss/articles/CBMipAFBVV95cUxNOWNiT2R6ZDF4S2QyWmZGbEhvOFIwbllXR09qZzR0SEl3ZGJOb2xhT2EtdGo3TEpNVElRZXUzbElWYjlpZFpzUi1ja3FDZG53N2JPR2NrcFgxYy1MeGpfNXg2dzhjbDdOcTdIZERTUHlFRnpSQ0RkVFl1TEo0Z2Roa1otalBaNm5PRllRZ1dWc1hQRVh2N1pvUHRQUWs3LVg1eGdqVQ?oc=5\" target=\"_blank\">Why benchmarking clinical LLMs from OpenEvidence, Doximity is complicated</a>&nbsp;&nbsp;<font color=\"#6f6f6f\">STAT</font>","source":"STAT","date":"2026-07-29T13:33:25+00:00","content":"","confidence":0.0,"diagnostics_url":"/api/diagnose?url=https%3A//www.statnews.com/2026/07/29/benchmarking-clinical-chatbots-openevidence-doximity-ai-prognosis/","quality_bucket":"failed","failure_kind":"http","retryable":false,"quality_reason":"Low confidence: HTTP extraction failure (http_403).","quality_profile":{"profile_version":"extraction_quality.v2","bucket":"failed","confidence":0.0,"failure_kind":"http","retryable":false,"retry_after_attempts":0,"reason":"Low confidence: HTTP extraction failure (http_403).","operator_guidance":{"severity":"action_required","recommended_action":"manual_review","next_step":"Open diagnostics and review the original source manually before use.","operator_label":"Manual review","can_retry":false,"can_use_summary":false,"diagnostics_required":true},"content_depth":{"contract_version":"content_depth.v1","category":"summary_only","label":"Summary only","has_full_text":false,"has_summary":true,"content_length":0,"summary_length":417,"usable_text_length":417,"source_field":"summary"},"legacy_collapsed":false,"signals":{"extract_state":"failed","extract_error":"http_403","extract_retries":2,"content_length":0,"summary_length":417}},"tags":[],"format_contract_version":"news_item_formats.v1"}}}