mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
test(ocr): cover official provider response shapes
This commit is contained in:
parent
39cec3ef4e
commit
189a96fdb6
4 changed files with 110 additions and 11 deletions
|
|
@ -155,19 +155,38 @@ mod tests {
|
|||
fn response_normalizes_markdown_images_blocks_and_billed_pages() {
|
||||
let response = serde_json::from_value(json!({
|
||||
"pages": [
|
||||
{"index": 4, "markdown": {"content": "receipt", "images": [{"id":"image", "bounding_box":{"top_left_x":1}, "description":"scan"}]}},
|
||||
{"blocks": [{"type":"text","text":"total"}]}
|
||||
{
|
||||
"type":"markdown",
|
||||
"index":4,
|
||||
"markdown":{
|
||||
"content":"receipt",
|
||||
"images":[{
|
||||
"id":"image",
|
||||
"bounding_box":{"top_left_x":1,"bottom_right_x":48},
|
||||
"bounding_box_normalized":{"top_left_x":0.04,"bottom_right_x":0.15},
|
||||
"description":"scan",
|
||||
"category":"logo"
|
||||
}]
|
||||
}
|
||||
},
|
||||
{"type":"blocks","blocks":[{"type":"text","text":{"content":"total"}}]}
|
||||
],
|
||||
"meta": {"billed_units":{"pages":3}}
|
||||
})).unwrap();
|
||||
"meta":{"api_version":{"version":"2"},"billed_units":{"pages":3}}
|
||||
}))
|
||||
.unwrap();
|
||||
let normalized = transform_response("parse-v5.0", response).unwrap();
|
||||
assert_eq!(normalized.pages[0]["index"], 4);
|
||||
assert_eq!(normalized.pages[0]["markdown"], "receipt");
|
||||
assert_eq!(normalized.pages[0]["images"][0]["bbox"]["top_left_x"], 1);
|
||||
assert_eq!(
|
||||
normalized.pages[0]["images"][0]["bounding_box_normalized"]["bottom_right_x"],
|
||||
0.15
|
||||
);
|
||||
assert_eq!(normalized.pages[0]["images"][0]["description"], "scan");
|
||||
assert_eq!(normalized.pages[0]["images"][0]["category"], "logo");
|
||||
assert_eq!(normalized.pages[1]["index"], 1);
|
||||
assert_eq!(normalized.pages[1]["markdown"], "");
|
||||
assert_eq!(normalized.pages[1]["blocks"][0]["text"], "total");
|
||||
assert_eq!(normalized.pages[1]["blocks"][0]["text"]["content"], "total");
|
||||
assert_eq!(normalized.usage_info.unwrap()["pages_processed"], 3);
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -114,6 +114,7 @@ mod tests {
|
|||
#[rstest]
|
||||
#[case("table_format", json!("html"))]
|
||||
#[case("confidence_scores_granularity", json!("word"))]
|
||||
#[case("confidence_scores_granularity", json!("block"))]
|
||||
#[case("document_annotation_prompt", json!("extract"))]
|
||||
#[case("include_blocks", json!(true))]
|
||||
#[case("id", json!("req-123"))]
|
||||
|
|
@ -197,8 +198,16 @@ mod tests {
|
|||
#[rstest]
|
||||
fn transform_ocr_response_preserves_blocks_and_confidence_scores() {
|
||||
let response: MistralOcrResponse = serde_json::from_value(json!({
|
||||
"pages":[{"index":0,"markdown":"hello","blocks":[{"type":"title"}],"confidence_scores":{"mean":0.99}}],
|
||||
"pages":[{
|
||||
"index":0,
|
||||
"markdown":"hello",
|
||||
"images":[{"id":"img-0","image_base64":"data:image/png;base64,AA=="}],
|
||||
"dimensions":{"width":612,"height":792,"dpi":72},
|
||||
"blocks":[{"type":"title","bbox":{"x":1},"confidence_scores":{"mean":0.98}}],
|
||||
"confidence_scores":{"average_page_confidence_score":0.99,"minimum_page_confidence_score":0.97}
|
||||
}],
|
||||
"model":"returned-model",
|
||||
"document_annotation":"{\"language\":\"en\"}",
|
||||
"usage_info":{"pages_processed":1}
|
||||
}))
|
||||
.unwrap();
|
||||
|
|
@ -206,7 +215,20 @@ mod tests {
|
|||
.unwrap()
|
||||
.into_json();
|
||||
assert_eq!(result["pages"][0]["blocks"][0]["type"], "title");
|
||||
assert_eq!(result["pages"][0]["confidence_scores"]["mean"], 0.99);
|
||||
assert_eq!(result["pages"][0]["blocks"][0]["bbox"]["x"], 1);
|
||||
assert_eq!(
|
||||
result["pages"][0]["blocks"][0]["confidence_scores"]["mean"],
|
||||
0.98
|
||||
);
|
||||
assert_eq!(
|
||||
result["pages"][0]["confidence_scores"]["average_page_confidence_score"],
|
||||
0.99
|
||||
);
|
||||
assert_eq!(result["pages"][0]["images"][0]["id"], "img-0");
|
||||
assert_eq!(result["pages"][0]["dimensions"]["dpi"], 72);
|
||||
assert_eq!(result["model"], "returned-model");
|
||||
assert_eq!(result["document_annotation"], "{\"language\":\"en\"}");
|
||||
assert_eq!(result["usage_info"]["pages_processed"], 1);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
|
|
|
|||
|
|
@ -177,6 +177,47 @@ mod tests {
|
|||
use super::*;
|
||||
use serde_json::json;
|
||||
|
||||
#[test]
|
||||
fn document_variants_preserve_provider_fields_when_rewriting_sources() {
|
||||
for (value, original, replacement, expected) in [
|
||||
(
|
||||
json!({
|
||||
"type":"document_url",
|
||||
"document_url":"https://example.com/input.pdf",
|
||||
"document_name":"input.pdf"
|
||||
}),
|
||||
"https://example.com/input.pdf",
|
||||
"data:application/pdf;base64,AA==",
|
||||
json!({
|
||||
"type":"document_url",
|
||||
"document_url":"data:application/pdf;base64,AA==",
|
||||
"document_name":"input.pdf"
|
||||
}),
|
||||
),
|
||||
(
|
||||
json!({
|
||||
"type":"image_url",
|
||||
"image_url":"https://example.com/input.png",
|
||||
"detail":"high"
|
||||
}),
|
||||
"https://example.com/input.png",
|
||||
"data:image/png;base64,AA==",
|
||||
json!({
|
||||
"type":"image_url",
|
||||
"image_url":"data:image/png;base64,AA==",
|
||||
"detail":"high"
|
||||
}),
|
||||
),
|
||||
] {
|
||||
let document: OcrDocument = serde_json::from_value(value).unwrap();
|
||||
assert_eq!(document.source(), original);
|
||||
assert_eq!(
|
||||
serde_json::to_value(document.with_source(replacement.into())).unwrap(),
|
||||
expected
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn response_serialization_flattens_extra_fields_and_omits_absent_native_response() {
|
||||
let response = LiteLLMOcrResponse {
|
||||
|
|
|
|||
|
|
@ -184,9 +184,16 @@ async fn rejects_invalid_document_sources_before_network(#[case] source: &str) {
|
|||
fn response_normalization_groups_blocks_and_distinguishes_null_result() {
|
||||
use crate::ocr::codecs::reducto::{ReductoResponse, transform_ocr_response};
|
||||
|
||||
let raw = json!({"usage":{"num_pages":"2","credits":"3"},"result":{"chunks":[
|
||||
{"blocks":[{"content":"B","bbox":{"page":2},"kind":"table"}]},
|
||||
{"blocks":[{"content":"A","bbox":{"page":1},"kind":"text"},{"content":"C","bbox":{"page":1}}]}
|
||||
let raw = json!({"usage":{"num_pages":"2","credits":"3"},"result":{"type":"full","chunks":[
|
||||
{"blocks":[{
|
||||
"type":"Table",
|
||||
"content":"B",
|
||||
"bbox":{"left":0.1,"top":0.2,"width":0.8,"height":0.3,"page":2,"original_page":4},
|
||||
"confidence":"high",
|
||||
"granular_confidence":{"parse_confidence":0.95,"extract_confidence":null},
|
||||
"image_url":null
|
||||
}]},
|
||||
{"blocks":[{"content":"A","bbox":{"page":1},"type":"Text"},{"content":"C","bbox":{"page":1}}]}
|
||||
]}});
|
||||
let response: ReductoResponse = serde_json::from_value(raw).unwrap();
|
||||
let normalized = transform_ocr_response("parse-v3", response)
|
||||
|
|
@ -194,7 +201,17 @@ fn response_normalization_groups_blocks_and_distinguishes_null_result() {
|
|||
.into_json();
|
||||
assert_eq!(normalized["pages"][0]["markdown"], "A\n\nC");
|
||||
assert_eq!(normalized["pages"][1]["markdown"], "B");
|
||||
assert_eq!(normalized["pages"][1]["blocks"][0]["kind"], "table");
|
||||
assert_eq!(normalized["pages"][1]["blocks"][0]["type"], "Table");
|
||||
assert_eq!(
|
||||
normalized["pages"][1]["blocks"][0]["bbox"],
|
||||
json!({"left":0.1,"top":0.2,"width":0.8,"height":0.3,"page":2,"original_page":4})
|
||||
);
|
||||
assert_eq!(normalized["pages"][1]["blocks"][0]["confidence"], "high");
|
||||
assert_eq!(
|
||||
normalized["pages"][1]["blocks"][0]["granular_confidence"]["parse_confidence"],
|
||||
0.95
|
||||
);
|
||||
assert!(normalized["pages"][1]["blocks"][0]["image_url"].is_null());
|
||||
assert_eq!(normalized["usage_info"]["pages_processed"], 2);
|
||||
assert_eq!(normalized["usage_info"]["credits"], 3.0);
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue