{"dcterms:modified":"2026-08-25","dcterms:creator":"DataverseUA","@type":"ore:ResourceMap","schema:additionalType":"Dataverse OREMap Format v1.0.0","dvcore:generatedBy":{"@type":"schema:SoftwareApplication","schema:name":"Dataverse","schema:version":"6.2","schema:url":"https://github.com/iqss/dataverse"},"@id":"https://opendata.nas.gov.ua/api/datasets/export?exporter=OAI_ORE&persistentId=https://doi.org/10.48788/DVUA/BM3ACV","ore:describes":{"citation:dsDescription":{"citation:dsDescriptionValue":"The dataset contains 355 classified advertisements organized into 15 semantic categories and represented as structured JSON objects for supervised multi-class text classification. Each advertisement includes a unique identifier, category identifier and title, advertisement title, full advertisement text, and an LLM-assisted summary. The accompanying category data provide category identifiers and titles together with category-level bag-of-words (BOW) and TF-IDF representations derived from the advertisement corpus.\nThe corpus consists predominantly of Ukrainian-language advertisements and includes naturally occurring mixed Ukrainian–Russian content. The texts preserve characteristics of real-world advertisements, including spelling variations, colloquial language, repetitions, commercial information, and stylistic variability. The dataset covers multiple thematic domains, including furniture, commercial premises and rentals, cosmetics, perfumery, healthcare and beauty products and services, medical products, and equipment.\nThe dataset is intended for research and educational purposes and can be used for supervised text classification, evaluation and benchmarking of machine learning and large language model (LLM)-based classifiers, natural language processing research, feature engineering, and comparative evaluation of text classification methods."},"author":{"citation:authorName":"Zharkov, Dmytro","citation:authorAffiliation":"V.M. Glushkov Institute of Cybernetics of the NAS of Ukraine","authorIdentifierScheme":"ORCID","authorIdentifier":"0009-0006-2700-5313"},"citation:datasetContact":{"citation:datasetContactName":"Zharkov, Dmytro","citation:datasetContactAffiliation":"V.M. Glushkov Institute of Cybernetics of the NAS of Ukraine","citation:datasetContactEmail":"d.zharkov@nas.gov.ua"},"dateOfDeposit":"2026-04-24","citation:depositor":"Zharkov, Dmytro","subject":"Computer and Information Science","title":"Classified Adverts Collection","citation:producer":{"citation:producerName":"V.M. Glushkov Institute of Cybernetics of the NAS of Ukraine","citation:producerAffiliation":"National Academy of Sciences of Ukraine","citation:producerAbbreviation":"IC NASU","citation:producerURL":"https://incyb.kiev.ua/en"},"citation:keyword":[{"citation:keywordValue":"text classification","citation:keywordVocabulary":"ACM Computing Classification System (CCS)"},{"citation:keywordValue":"machine learning"},{"citation:keywordValue":"Ukrainian language"},{"citation:keywordValue":"NLP"},{"citation:keywordValue":"document classification"},{"citation:keywordValue":"large language models"}],"citation:characteristicOfSources":"The corpus comprises 355 classified advertisement records distributed across 15 semantic categories. The textual content is predominantly Ukrainian and includes mixed Ukrainian–Russian language features characteristic of real-world user-generated advertisements. Records preserve spelling variations, colloquial expressions, commercial information, repetitions, and stylistic variability. Each advertisement is associated with a single category label.","citation:subtitle":"Dataset for Large-Scale Text Classification of Ukrainian Classified Advertisements","language":"Ukrainian","dataSources":"Classified advertisement texts organized into semantic categories and represented as structured JSON data for text classification and natural language processing research.","citation:accessToSources":"The dataset is accompanied by a README file describing the dataset purpose, JSON file structure, advertisement and category objects, category labeling, dataset characteristics, and recommended uses. The dataset includes adverts.json with 355 structured advertisement records and categories.json with 15 category definitions and corresponding bag-of-words (BOW) and TF-IDF representations.","alternativeTitle":"Ukrainian Classified Advertisements Dataset","@id":"https://doi.org/10.48788/DVUA/BM3ACV","@type":["ore:Aggregation","schema:Dataset"],"schema:version":"1.0","schema:name":"Classified Adverts Collection","schema:dateModified":"2026-08-25 09:28:45.353","schema:datePublished":"2026-08-25","schema:creativeWorkStatus":"RELEASED","schema:license":"http://creativecommons.org/licenses/by/4.0","dvcore:fileTermsOfAccess":{"dvcore:fileRequestAccess":true},"schema:includedInDataCatalog":"DataverseUA","schema:isPartOf":{"schema:name":"V.M. Glushkov Institute of Cybernetics of the NAS of Ukraine","@id":"https://opendata.nas.gov.ua/dataverse/incyb","schema:description":"Main directions of scientific research of the V.M. Glushkov Institute of Cybernetics are as follows: \n\n  - the development of a general theory and methods of systems analysis, mathematical modeling, optimization, and artificial intelligence; \n\n  - the development of a general theory of control and methods and means for the construction of intelligent control systems of different levels and destinations;\n\n  - the development of new information technologies and intelligent systems.","schema:isPartOf":{"schema:name":"DataverseUA","@id":"https://opendata.nas.gov.ua/dataverse/dataverseua","schema:description":"Dataverse UA is a national research data repository designed to support open science and implement the FAIR principles (Findable, Accessible, Interoperable, Reusable) in Ukraine. Its mission is to ensure long-term preservation, publication, and reuse of research outputs in standardized formats. The platform enables researchers, educators, graduate students, and institutions to store datasets, metadata, supporting documents, and code — all with persistent identifiers (DOIs), Creative Commons licenses, and ORCID integration.\nDataverse UA supports both Ukrainian and English, allows the creation of thematic collections by discipline or institution, and complies with the standards of the National Academy of Sciences of Ukraine and international best practices. It is a tool for enhancing transparency, reliability, and reproducibility in research, while connecting Ukrainian science to global open data ecosystems."}},"ore:aggregates":[{"schema:description":"Documentation describing the dataset.","schema:name":"README.txt","dvcore:restricted":false,"schema:version":1,"dvcore:datasetVersionId":102,"@id":"https://opendata.nas.gov.ua/file.xhtml?fileId=1486","schema:sameAs":"https://opendata.nas.gov.ua/api/access/datafile/1486","@type":"ore:AggregatedResource","schema:fileFormat":"text/plain","dvcore:filesize":2699,"dvcore:storageIdentifier":"local://19fc9403d98-9625474a8098","dvcore:rootDataFileId":-1,"dvcore:checksum":{"@type":"MD5","@value":"57e268c225c30311d07263c377977515"}},{"schema:description":"A structured collection of advertisements","schema:name":"adverts.json","dvcore:restricted":false,"schema:version":1,"dvcore:datasetVersionId":102,"@id":"https://opendata.nas.gov.ua/file.xhtml?fileId=1146","schema:sameAs":"https://opendata.nas.gov.ua/api/access/datafile/1146","@type":"ore:AggregatedResource","schema:fileFormat":"application/json","dvcore:filesize":302450,"dvcore:storageIdentifier":"local://19dc059265b-b94a4f034d0d","dvcore:rootDataFileId":-1,"dvcore:checksum":{"@type":"MD5","@value":"f11b490a2522f527bb357197f5fb0e55"}},{"schema:description":"A set of target categories enriched with collections of characteristic keywords specific to each class","schema:name":"categories.json","dvcore:restricted":false,"schema:version":1,"dvcore:datasetVersionId":102,"@id":"https://opendata.nas.gov.ua/file.xhtml?fileId=1147","schema:sameAs":"https://opendata.nas.gov.ua/api/access/datafile/1147","@type":"ore:AggregatedResource","schema:fileFormat":"application/json","dvcore:filesize":19934,"dvcore:storageIdentifier":"local://19dc0596ab5-bd206382c5bd","dvcore:rootDataFileId":-1,"dvcore:checksum":{"@type":"MD5","@value":"46c2db615ab9e5e993ff0614b476da3b"}}],"schema:hasPart":["https://opendata.nas.gov.ua/file.xhtml?fileId=1486","https://opendata.nas.gov.ua/file.xhtml?fileId=1146","https://opendata.nas.gov.ua/file.xhtml?fileId=1147"]},"@context":{"alternativeTitle":"http://purl.org/dc/terms/alternative","author":"http://purl.org/dc/terms/creator","authorIdentifier":"http://purl.org/spar/datacite/AgentIdentifier","authorIdentifierScheme":"http://purl.org/spar/datacite/AgentIdentifierScheme","citation":"https://dataverse.org/schema/citation/","dataSources":"https://www.w3.org/TR/prov-o/#wasDerivedFrom","dateOfDeposit":"http://purl.org/dc/terms/dateSubmitted","dcterms":"http://purl.org/dc/terms/","dvcore":"https://dataverse.org/schema/core#","language":"http://purl.org/dc/terms/language","ore":"http://www.openarchives.org/ore/terms/","schema":"http://schema.org/","subject":"http://purl.org/dc/terms/subject","title":"http://purl.org/dc/terms/title"}}