{"id":"https://openalex.org/W6893480033","doi":"https://doi.org/10.5281/zenodo.15881917","title":"IndicDLP: A Foundational Dataset for Multi-Lingual and Multi-Domain Document Layout Parsing","display_name":"IndicDLP: A Foundational Dataset for Multi-Lingual and Multi-Domain Document Layout Parsing","publication_year":2025,"publication_date":"2025-07-14","ids":{"openalex":"https://openalex.org/W6893480033","doi":"https://doi.org/10.5281/zenodo.15881917"},"language":"en","primary_location":{"id":"doi:10.5281/zenodo.15881917","is_oa":true,"landing_page_url":"https://doi.org/10.5281/zenodo.15881917","pdf_url":null,"source":{"id":"https://openalex.org/S4306400562","display_name":"Zenodo (CERN European Organization for Nuclear Research)","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I67311998","host_organization_name":"European Organization for Nuclear Research","host_organization_lineage":["https://openalex.org/I67311998"],"host_organization_lineage_names":[],"type":"repository"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":null,"is_accepted":false,"is_published":false,"raw_source_name":null,"raw_type":"dataset"},"type":"dataset","indexed_in":["datacite"],"open_access":{"is_oa":true,"oa_status":"green","oa_url":"https://doi.org/10.5281/zenodo.15881917","any_repository_has_fulltext":true},"authorships":[{"author_position":"first","author":{"id":null,"display_name":"Nath, Oikantik","orcid":"https://orcid.org/0009-0005-5407-0455"},"institutions":[{"id":"https://openalex.org/I24676775","display_name":"Indian Institute of Technology Madras","ror":"https://ror.org/03v0r5n49","country_code":"IN","type":"education","lineage":["https://openalex.org/I24676775"]}],"countries":["IN"],"is_corresponding":false,"raw_author_name":"Nath, Oikantik","raw_affiliation_strings":["Indian Institute of Technology Madras"],"raw_orcid":"https://orcid.org/0009-0005-5407-0455","affiliations":[{"raw_affiliation_string":"Indian Institute of Technology Madras","institution_ids":["https://openalex.org/I24676775"]}]},{"author_position":"middle","author":{"id":null,"display_name":"Kukkala, Sahithi","orcid":"https://orcid.org/0009-0003-5483-2232"},"institutions":[{"id":"https://openalex.org/I64189192","display_name":"International Institute of Information Technology, Hyderabad","ror":"https://ror.org/05f11g639","country_code":"IN","type":"education","lineage":["https://openalex.org/I64189192"]}],"countries":["IN"],"is_corresponding":false,"raw_author_name":"Kukkala, Sahithi","raw_affiliation_strings":["International Institute of Information Technology, Hyderabad"],"raw_orcid":"https://orcid.org/0009-0003-5483-2232","affiliations":[{"raw_affiliation_string":"International Institute of Information Technology, Hyderabad","institution_ids":["https://openalex.org/I64189192"]}]},{"author_position":"middle","author":{"id":null,"display_name":"Khapra, Mitesh","orcid":"https://orcid.org/0009-0008-3687-9922"},"institutions":[{"id":"https://openalex.org/I24676775","display_name":"Indian Institute of Technology Madras","ror":"https://ror.org/03v0r5n49","country_code":"IN","type":"education","lineage":["https://openalex.org/I24676775"]}],"countries":["IN"],"is_corresponding":false,"raw_author_name":"Khapra, Mitesh","raw_affiliation_strings":["Indian Institute of Technology Madras"],"raw_orcid":"https://orcid.org/0009-0008-3687-9922","affiliations":[{"raw_affiliation_string":"Indian Institute of Technology Madras","institution_ids":["https://openalex.org/I24676775"]}]},{"author_position":"last","author":{"id":null,"display_name":"Sarvadevabhatla, Ravi Kiran","orcid":"https://orcid.org/0000-0003-4134-1154"},"institutions":[{"id":"https://openalex.org/I64189192","display_name":"International Institute of Information Technology, Hyderabad","ror":"https://ror.org/05f11g639","country_code":"IN","type":"education","lineage":["https://openalex.org/I64189192"]}],"countries":["IN"],"is_corresponding":false,"raw_author_name":"Sarvadevabhatla, Ravi Kiran","raw_affiliation_strings":["International Institute of Information Technology, Hyderabad"],"raw_orcid":"https://orcid.org/0000-0003-4134-1154","affiliations":[{"raw_affiliation_string":"International Institute of Information Technology, Hyderabad","institution_ids":["https://openalex.org/I64189192"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":4,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":0,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":null,"issue":null,"first_page":null,"last_page":null},"is_retracted":false,"is_paratext":false,"is_xpac":true,"primary_topic":null,"topics":[],"keywords":[{"id":"https://openalex.org/keywords/parsing","display_name":"Parsing","score":0.807200014591217},{"id":"https://openalex.org/keywords/document-layout-analysis","display_name":"Document layout analysis","score":0.6384999752044678},{"id":"https://openalex.org/keywords/information-extraction","display_name":"Information extraction","score":0.45890000462532043},{"id":"https://openalex.org/keywords/document-processing","display_name":"Document processing","score":0.39969998598098755},{"id":"https://openalex.org/keywords/historical-document","display_name":"Historical document","score":0.3499999940395355},{"id":"https://openalex.org/keywords/semantics","display_name":"Semantics (computer science)","score":0.34130001068115234},{"id":"https://openalex.org/keywords/range","display_name":"Range (aeronautics)","score":0.3328999876976013},{"id":"https://openalex.org/keywords/document-structure-description","display_name":"Document Structure Description","score":0.3271999955177307}],"concepts":[{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.8766000270843506},{"id":"https://openalex.org/C186644900","wikidata":"https://www.wikidata.org/wiki/Q194152","display_name":"Parsing","level":2,"score":0.807200014591217},{"id":"https://openalex.org/C72773152","wikidata":"https://www.wikidata.org/wiki/Q5287629","display_name":"Document layout analysis","level":3,"score":0.6384999752044678},{"id":"https://openalex.org/C204321447","wikidata":"https://www.wikidata.org/wiki/Q30642","display_name":"Natural language processing","level":1,"score":0.5171999931335449},{"id":"https://openalex.org/C23123220","wikidata":"https://www.wikidata.org/wiki/Q816826","display_name":"Information retrieval","level":1,"score":0.5162000060081482},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.46860000491142273},{"id":"https://openalex.org/C195807954","wikidata":"https://www.wikidata.org/wiki/Q1662562","display_name":"Information extraction","level":2,"score":0.45890000462532043},{"id":"https://openalex.org/C67905146","wikidata":"https://www.wikidata.org/wiki/Q5287646","display_name":"Document processing","level":2,"score":0.39969998598098755},{"id":"https://openalex.org/C2778371909","wikidata":"https://www.wikidata.org/wiki/Q3771738","display_name":"Historical document","level":2,"score":0.3499999940395355},{"id":"https://openalex.org/C184337299","wikidata":"https://www.wikidata.org/wiki/Q1437428","display_name":"Semantics (computer science)","level":2,"score":0.34130001068115234},{"id":"https://openalex.org/C204323151","wikidata":"https://www.wikidata.org/wiki/Q905424","display_name":"Range (aeronautics)","level":2,"score":0.3328999876976013},{"id":"https://openalex.org/C68699486","wikidata":"https://www.wikidata.org/wiki/Q265904","display_name":"Document Structure Description","level":3,"score":0.3271999955177307},{"id":"https://openalex.org/C2777737414","wikidata":"https://www.wikidata.org/wiki/Q4868296","display_name":"Font","level":2,"score":0.3075000047683716},{"id":"https://openalex.org/C188985296","wikidata":"https://www.wikidata.org/wiki/Q868954","display_name":"Page layout","level":2,"score":0.29989999532699585},{"id":"https://openalex.org/C137293760","wikidata":"https://www.wikidata.org/wiki/Q3621696","display_name":"Language model","level":2,"score":0.2784999907016754},{"id":"https://openalex.org/C2776207758","wikidata":"https://www.wikidata.org/wiki/Q5303302","display_name":"Downstream (manufacturing)","level":2,"score":0.27140000462532043},{"id":"https://openalex.org/C2780451532","wikidata":"https://www.wikidata.org/wiki/Q759676","display_name":"Task (project management)","level":2,"score":0.27070000767707825},{"id":"https://openalex.org/C137441365","wikidata":"https://www.wikidata.org/wiki/Q7981054","display_name":"Well-formed document","level":5,"score":0.26579999923706055},{"id":"https://openalex.org/C67277372","wikidata":"https://www.wikidata.org/wiki/Q7449085","display_name":"Semantic role labeling","level":3,"score":0.26190000772476196},{"id":"https://openalex.org/C2988504005","wikidata":"https://www.wikidata.org/wiki/Q379942","display_name":"Document image processing","level":4,"score":0.258899986743927},{"id":"https://openalex.org/C161156560","wikidata":"https://www.wikidata.org/wiki/Q1638872","display_name":"Document retrieval","level":2,"score":0.25380000472068787}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.5281/zenodo.15881917","is_oa":true,"landing_page_url":"https://doi.org/10.5281/zenodo.15881917","pdf_url":null,"source":{"id":"https://openalex.org/S4306400562","display_name":"Zenodo (CERN European Organization for Nuclear Research)","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I67311998","host_organization_name":"European Organization for Nuclear Research","host_organization_lineage":["https://openalex.org/I67311998"],"host_organization_lineage_names":[],"type":"repository"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":null,"is_accepted":false,"is_published":null,"raw_source_name":null,"raw_type":"dataset"}],"best_oa_location":{"id":"doi:10.5281/zenodo.15881917","is_oa":true,"landing_page_url":"https://doi.org/10.5281/zenodo.15881917","pdf_url":null,"source":{"id":"https://openalex.org/S4306400562","display_name":"Zenodo (CERN European Organization for Nuclear Research)","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I67311998","host_organization_name":"European Organization for Nuclear Research","host_organization_lineage":["https://openalex.org/I67311998"],"host_organization_lineage_names":[],"type":"repository"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":null,"is_accepted":false,"is_published":false,"raw_source_name":null,"raw_type":"dataset"},"sustainable_development_goals":[{"display_name":"Quality Education","score":0.6799142360687256,"id":"https://metadata.un.org/sdg/4"}],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":0,"referenced_works":[],"related_works":[],"abstract_inverted_index":{"IndicDLP":[0,2,77,115,116,133],"Dataset":[1],"is":[3,98],"a":[4,145],"large-scale,":[5],"foundational":[6],"dataset":[7,44,69,97],"created":[8,86],"to":[9,100,182],"advance":[10],"document":[11,22,48,110,141,152],"layout":[12,75,103,142],"parsing":[13,143],"in":[14,164],"multi-lingual":[15],"and":[16,28,41,66,73,81,109,127,151,154,176,192],"multi-domain":[17],"settings.":[18],"It":[19],"comprises":[20],"119,806":[21],"images":[23],"covering":[24],"11":[25],"Indic":[26,149],"languages":[27,150],"English:":[29],"Assamese,":[30],"Bengali,":[31],"English,":[32],"Gujarati,":[33],"Hindi,":[34],"Kannada,":[35],"Malayalam,":[36],"Marathi,":[37],"Odia,":[38],"Punjabi,":[39],"Tamil,":[40],"Telugu.":[42],"The":[43,68,96],"spans":[45],"12":[46],"diverse":[47,106],"categories,":[49],"including":[50],"Novels,":[51],"Textbooks,":[52],"Magazines,":[53],"Acts":[54],"&":[55],"Rules,":[56],"Research":[57],"Papers,":[58,63],"Manuals,":[59],"Brochures,":[60],"Syllabi,":[61],"Question":[62],"Notices,":[64],"Forms,":[65],"Newspapers.":[67],"contains":[70],"42":[71,160],"physical":[72],"logical":[74],"classes.":[76],"includes":[78],"both":[79,174],"digitally-born":[80,177],"scanned":[82,175],"documents,":[83],"with":[84],"annotations":[85],"using":[87],"Shoonya,":[88],"an":[89],"open-source":[90],"tool":[91],"built":[92],"on":[93,131,173],"Label":[94],"Studio.":[95],"curated":[99],"support":[101],"robust":[102,140],"understanding":[104],"across":[105,144],"scripts,":[107],"domains,":[108],"types.":[111],"Project":[112],"Page":[113],":":[114],"Model":[117],"Checkpoints":[118],"We":[119],"provide":[120],"3":[121],"model":[122],"checkpoints":[123,168],"\u2014":[124,129],"YOLOv10x,":[125],"DocLayout-YOLO,":[126],"RoDLA":[128],"finetuned":[130],"the":[132,165],"dataset.":[134,166],"These":[135,167],"models":[136],"are":[137,155,180],"optimized":[138],"for":[139,184,190,197],"wide":[146],"range":[147],"of":[148,157],"types,":[153],"capable":[156],"detecting":[158],"all":[159],"region":[161],"labels":[162],"defined":[163],"have":[169],"demonstrated":[170],"strong":[171,188],"performance":[172],"documents.":[178],"They":[179],"ready":[181],"use":[183],"inference,":[185],"serve":[186],"as":[187,201],"baselines":[189],"benchmarking,":[191],"can":[193],"be":[194],"further":[195],"fine-tuned":[196],"downstream":[198],"tasks":[199],"such":[200],"structure":[202],"extraction":[203],"or":[204],"semantic":[205],"tagging.":[206]},"counts_by_year":[],"updated_date":"2026-06-11T09:08:48.828518","created_date":"2025-10-10T00:00:00"}
