[
  {
    "id": "vietmix",
    "title": "VietMix",
    "summary": "Vietnamese-English Code-Mixed Machine Translation Corpus",
    "type": "Machine Translation",
    "size": "Updating...",
    "sizeBytes": 18300000000,
    "entries": "10,254",
    "entriesLabel": "posts",
    "paper": "https://aclanthology.org/2026.eacl-long.342/"
  },
  {
    "id": "vimedcss",
    "title": "ViMedCSS",
    "summary": "A Vietnamese Medical Code-Switching Speech Dataset",
    "type": "Medical-Speech",
    "size": "Updating...",
    "sizeBytes": 18300000000,
    "entries": "15,818",
    "entriesLabel": "reviews",
    "paper": "https://arxiv.org/abs/2602.12911"
  },
  {
    "id": "vnlawqc",
    "title": "VNLawQC",
    "summary": "Natural Language Inference for Vietnamese Legal Texts",
    "type": "NLI/RTE",
    "size": "Updating...",
    "sizeBytes": 1200000000,
    "entries": "224,008",
    "entriesLabel": "pairs",
    "paper": "https://aclanthology.org/2025.naacl-short.12"
  },
  {
    "id": "vianli",
    "title": "ViANLI",
    "summary": "Adversarial Natural Language Inference in Vietnamese",
    "type": "NLI/RTE",
    "size": "Updating...",
    "sizeBytes": 1200000000,
    "entries": "10,000",
    "entriesLabel": "pairs",
    "paper": "https://arxiv.org/abs/2406.17716"
  },
  {
    "id": "vimrhp",
    "title": "ViMRHP",
    "summary": "Multimodal Review Helpfulness Prediction dataset with Vietnamese product reviews and images.",
    "type": "Review-Helpfulness",
    "size": "Updating...",
    "sizeBytes": 1200000000,
    "entries": "46,000",
    "entriesLabel": "reviews",
    "paper": "https://arxiv.org/abs/2505.07416"
  },
  {
    "id": "vlqa",
    "title": "VLQA",
    "summary": "The First Comprehensive, Large, and High-Quality Vietnamese Dataset for Legal Question Answering",
    "type": "Legal QA",
    "size": "Updating...",
    "sizeBytes": 450000000,
    "entries": "3,129",
    "entriesLabel": "{question, articles, answer} triples",
    "paper": "https://arxiv.org/abs/2507.19995"
  },
  {
    "id": "vitosa",
    "title": "ViToSA",
    "summary": "Vietnamese Toxic Spans Audio Dataset",
    "type": "Span Detection",
    "size": "Updating...",
    "sizeBytes": 12800000000,
    "entries": "12,802",
    "entriesLabel": "docs",
    "paper": "https://www.isca-archive.org/interspeech_2025/do25b_interspeech.pdf"
  },
  {
    "id": "alqac2023",
    "title": "ALQAC 2023",
    "summary": "Multi-domain social media reviews with aspect-based labels.",
    "type": "Legal QA",
    "size": "Updating...",
    "sizeBytes": 1200000000,
    "entries": "450,000",
    "entriesLabel": "reviews",
    "paper": "https://ieeexplore.ieee.org/document/10299527"
  },
  {
    "id": "viocrvqa",
    "title": "ViOCRVQA",
    "summary": "Visual question answering by understanding Vietnamese text in images",
    "type": "Book VQA",
    "size": "Updating...",
    "sizeBytes": 800000000,
    "entries": "25,000",
    "entriesLabel": "facts",
    "paper": "https://arxiv.org/abs/2404.18397"
  },
  {
    "id": "vifactcheck",
    "title": "ViFactCheck",
    "summary": "Fact-checking dataset for Vietnamese claims.",
    "type": "Fact-Checking",
    "size": "Updating...",
    "sizeBytes": 680000000,
    "entries": "7,000",
    "entriesLabel": "pairs",
    "paper": "https://arxiv.org/abs/2412.15308"
  },
  {
    "id": "vilexnorm",
    "title": "ViLexNorm",
    "summary": "A Lexical Normalization Corpus for Vietnamese Social Media Text",
    "type": "Normalization",
    "size": "Updating...",
    "sizeBytes": 920000000,
    "entries": "310,000",
    "entriesLabel": "comments",
    "paper": "https://aclanthology.org/2024.eacl-long.85/"
  },
  {
    "id": "openvivqa",
    "title": "OpenViVQA",
    "summary": "Visual question answering in Vietnamese with open-domain images",
    "type": "VQA",
    "size": "Updating...",
    "sizeBytes": 1600000000,
    "entries": "37,000",
    "entriesLabel": "QA pairs",
    "paper": "https://arxiv.org/abs/2305.04183"
  },
  {
    "id": "viglue",
    "title": "ViGLUE",
    "summary": "A Vietnamese General Language Understanding Evaluation Benchmark",
    "type": "NLU",
    "size": "Updating...",
    "sizeBytes": 4100000000,
    "entries": "Updating...",
    "entriesLabel": "samples",
    "paper": "https://aclanthology.org/2024.findings-naacl.261/"
  },
  {
    "id": "bv-frd",
    "title": "BV-FRD",
    "summary": "A novel multimodal Vietnamese English dataset focused on Vietnamese food review videos",
    "type": "Food Review",
    "size": "Updating...",
    "sizeBytes": 6400000000,
    "entries": "2,020",
    "entriesLabel": "samples",
    "paper": "https://aclanthology.org/2025.paclic-1.15/"
  },
  {
    "id": "vinli",
    "title": "ViNLI",
    "summary": "Open-Domain Natural Language Inference in Vietnamese",
    "type": "NLI/RTE",
    "size": "Updating...",
    "sizeBytes": 2300000000,
    "entries": "30,000",
    "entriesLabel": "pairs",
    "paper": "https://aclanthology.org/2022.coling-1.339/"
  },
  {
    "id": "updating",
    "title": "Updating...",
    "summary": "Updating...",
    "type": "Updating...",
    "size": "Updating...",
    "sizeBytes": 4100000000,
    "entries": "Updating...",
    "entriesLabel": "samples",
    "paper": "https://example.com/papers/samples"
  }
]