feat: leverage new list modeling, capture default markers (#1856)

* chore: update docling-core & regenerate test data

Signed-off-by: Panos Vagenas <pva@zurich.ibm.com>

* update backends to leverage new list modeling

Signed-off-by: Panos Vagenas <pva@zurich.ibm.com>

* repin docling-core

Signed-off-by: Panos Vagenas <pva@zurich.ibm.com>

* ensure availability of latest docling-core API

Signed-off-by: Panos Vagenas <pva@zurich.ibm.com>

---------

Signed-off-by: Panos Vagenas <pva@zurich.ibm.com>
This commit is contained in:
Panos Vagenas
2025-06-27 16:37:15 +02:00
committed by GitHub
parent e79e4f0ab6
commit 0533da1923
90 changed files with 2252 additions and 2240 deletions

View File

@@ -1,6 +1,6 @@
{
"schema_name": "DoclingDocument",
"version": "1.4.0",
"version": "1.5.0",
"name": "redp5110_sampled",
"origin": {
"mimetype": "application/pdf",
@@ -1295,7 +1295,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/15",
@@ -1326,7 +1326,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/16",
@@ -1357,7 +1357,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/17",
@@ -1388,7 +1388,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/18",
@@ -1683,7 +1683,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/28",
@@ -1714,7 +1714,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/29",
@@ -1745,7 +1745,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/30",
@@ -1776,7 +1776,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/31",
@@ -1807,7 +1807,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/32",
@@ -1838,7 +1838,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/33",
@@ -1869,7 +1869,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/34",
@@ -1900,7 +1900,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/35",
@@ -1931,7 +1931,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/36",
@@ -2400,7 +2400,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/52",
@@ -2431,7 +2431,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/53",
@@ -2462,7 +2462,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/54",
@@ -2668,7 +2668,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/61",
@@ -2699,7 +2699,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/62",
@@ -2759,7 +2759,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/64",
@@ -3344,7 +3344,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/84",
@@ -3375,7 +3375,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/85",
@@ -3406,7 +3406,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/86",
@@ -5992,7 +5992,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/175",
@@ -6023,7 +6023,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/176",
@@ -6054,7 +6054,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/177",
@@ -6085,7 +6085,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/178",
@@ -6116,7 +6116,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/179",
@@ -6787,7 +6787,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/202",
@@ -6818,7 +6818,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/203",
@@ -6849,7 +6849,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/204",
@@ -7064,7 +7064,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/211",
@@ -7095,7 +7095,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/212",
@@ -7126,7 +7126,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/213",
@@ -7157,7 +7157,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/214",
@@ -7188,7 +7188,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/215",
@@ -7219,7 +7219,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/216",
@@ -7379,7 +7379,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/221",
@@ -7498,7 +7498,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/225",
@@ -7559,7 +7559,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/227",
@@ -7590,7 +7590,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/228",
@@ -7737,7 +7737,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/233",
@@ -7855,7 +7855,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/237",
@@ -7915,7 +7915,7 @@
"formatting": null,
"hyperlink": null,
"enumerated": false,
"marker": "-"
"marker": ""
},
{
"self_ref": "#/texts/239",