{
  "_schema": "https://data.nist.gov/od/dm/nerdm-schema/v0.7#",
  "@context": [
    "https://data.nist.gov/od/dm/nerdm-pub-context.jsonld",
    {
      "@base": "ark:/88434/mds2-4134"
    }
  ],
  "@type": [
    "nrdp:DataPublication",
    "nrdp:PublicDataResource",
    "dcat:Dataset"
  ],
  "_extensionSchemas": [
    "https://data.nist.gov/od/dm/nerdm-schema/pub/v0.7#/definitions/PublicDataResource"
  ],
  "@id": "ark:/88434/mds2-4134",
  "ediid": "ark:/88434/mds2-4134",
  "version": "1.0.0",
  "doi": "doi:10.18434/mds2-4134",
  "title": "TOVA: Topic Visualization & Analysis",
  "contactPoint": {
    "fn": "Juan Fung",
    "hasEmail": "mailto:juan.fung@nist.gov"
  },
  "modified": "2026-06-10",
  "status": "available",
  "landingPage": "https://data.nist.gov/od/id/mds2-4134",
  "description": [
    "TOVA is a topic modeling platform with a plug-in architecture, supporting training and inference via CLI and web interface.",
    "TOVA is a highly extensible topic modeling platform that supports both traditional and LLM-based models through a unified CLI and interactive web interface. It provides a comprehensive end-to-end workflow, allowing users to train models with full hyperparameter control, enrich topics with LLM-generated labels, and evaluate results using an interactive dashboard rich with visualizations and metrics. Additionally, TOVA offers robust features for running inference on new documents, built-in uncertainty quantification to estimate document-topic probabilities, and a flexible plug-in architecture that makes it easy to integrate new model classes."
  ],
  "keyword": [
    "natural language processing",
    "topic model",
    "artificial intelligence",
    "large language model",
    "content analysis",
    "grounded theory",
    "unstructured text"
  ],
  "topic": [
    {
      "@type": "Concept",
      "scheme": "https://data.nist.gov/od/dm/nist-themes/v1.1",
      "tag": "Information Technology: Computational science"
    },
    {
      "@type": "Concept",
      "scheme": "https://data.nist.gov/od/dm/nist-themes/v1.1",
      "tag": "Information Technology: Software research"
    }
  ],
  "accessLevel": "public",
  "license": "https://www.nist.gov/open/license",
  "publisher": {
    "name": "National Institute of Standards and Technology",
    "@type": "org:Organization"
  },
  "language": [
    "en"
  ],
  "bureauCode": [
    "006:55"
  ],
  "programCode": [
    "006:052"
  ],
  "_editStatus": "done",
  "theme": [
    "Information Technology: Computational science",
    "Information Technology: Software research"
  ],
  "references": [
    {
      "@type": [
        "npg:Document"
      ],
      "@id": "#ref:10.18653/v1/2024.eacl-long.51",
      "refType": "IsSupplementTo",
      "location": "https://doi.org/10.18653/v1/2024.eacl-long.51",
      "_extensionSchemas": [
        "https://data.nist.gov/od/dm/nerdm-schema/bib/v0.7#/definitions/DCiteReference"
      ],
      "title": "Improving the TENOR of Labeling: Re-evaluating Topic Models for Content Analysis",
      "issued": "2024",
      "citation": "Li, Z., Mao, A., Stephens, D., Goel, P., Walpole, E., Dima, A., Fung, J., & Boyd-Graber, J. (2024). Improving the TENOR of Labeling: Re-evaluating Topic Models for Content Analysis. Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers), 840\u00e2\u0080\u0093859. https://doi.org/10.18653/v1/2024.eacl-long.51\n"
    },
    {
      "@type": [
        "npg:Document"
      ],
      "@id": "#ref:10.18653/v1/2025.acl-long.375",
      "refType": "IsSupplementTo",
      "location": "https://doi.org/10.18653/v1/2025.acl-long.375",
      "_extensionSchemas": [
        "https://data.nist.gov/od/dm/nerdm-schema/bib/v0.7#/definitions/DCiteReference"
      ],
      "title": "Large Language Models Struggle to Describe the Haystack without Human Help: A Social Science-Inspired Evaluation of Topic Models",
      "issued": "2025",
      "citation": "Li, Z., Calvo-Bartolom\u00c3\u00a9, L., Hoyle, A. M., Xu, P., Stephens, D. K., Fung, J. F., Dima, A., & Boyd-Graber, J. L. (2025). Large Language Models Struggle to Describe the Haystack without Human Help: A Social Science-Inspired Evaluation of Topic Models. Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 7583\u00e2\u0080\u00937604. https://doi.org/10.18653/v1/2025.acl-long.375\n"
    }
  ],
  "components": [
    {
      "accessURL": "https://github.com/usnistgov/TOVA",
      "format": {
        "description": "code repo"
      },
      "description": "NIST GitHub repository for TOVA",
      "title": "TOVA.git",
      "@type": [
        "nrdp:AccessPage",
        "dcat:Distribution"
      ],
      "@id": "#usnistgov/TOVA",
      "_extensionSchemas": [
        "https://data.nist.gov/od/dm/nerdm-schema/pub/v0.7#/definitions/AccessPage"
      ]
    },
    {
      "@id": "cmps/readme.md",
      "@type": [
        "nrdp:DataFile",
        "nrdp:DownloadableFile",
        "dcat:Distribution"
      ],
      "_extensionSchemas": [
        "https://data.nist.gov/od/dm/nerdm-schema/pub/v0.7#/definitions/DataFile"
      ],
      "filepath": "readme.md",
      "downloadURL": "https://data.nist.gov/od/ds/mds2-4134/readme.md",
      "mediaType": "text/markdown",
      "description": "Describes purpose of TOVA, functionality, setup and configuration, and troubleshooting",
      "title": "README",
      "size": 13524,
      "checksum": {
        "hash": "725c92493e300baf0fa03e6679c2565b06ca0d3dd866fb6c8f209670fd69bb05",
        "algorithm": {
          "tag": "sha256",
          "@type": "Thing"
        }
      },
      "format": {
        "description": "markdown"
      }
    }
  ],
  "authors": [
    {
      "familyName": "Stephens",
      "fn": "Daniel Kofi Stephens",
      "givenName": "Daniel",
      "middleName": "Kofi",
      "affiliation": [
        {
          "title": "Morgan State University",
          "@type": "org:Organization"
        }
      ],
      "orcid": "",
      "@type": "foaf:Person"
    },
    {
      "familyName": "Calvo Bartolome",
      "fn": "Lorena  Calvo Bartolome",
      "givenName": "Lorena",
      "middleName": "",
      "affiliation": [
        {
          "title": "Universidad Carlos III Madrid",
          "@type": "org:Organization"
        }
      ],
      "orcid": "",
      "@type": "foaf:Person"
    },
    {
      "familyName": "Li",
      "fn": "Zongxia  Li",
      "givenName": "Zongxia",
      "middleName": "",
      "affiliation": [
        {
          "title": "University of Maryland",
          "@type": "org:Organization"
        }
      ],
      "orcid": "",
      "@type": "foaf:Person"
    },
    {
      "familyName": "Fung",
      "fn": "Juan Francisco Fung",
      "givenName": "Juan",
      "middleName": "Francisco",
      "affiliation": [
        {
          "title": "National Institute of Standards and Technology",
          "@type": "org:Organization",
          "@id": "ror:05xpvk416"
        }
      ],
      "orcid": "0000-0002-0820-787X",
      "@type": "foaf:Person"
    },
    {
      "familyName": "Dima",
      "fn": "Alden  Dima",
      "givenName": "Alden",
      "middleName": "",
      "affiliation": [
        {
          "title": "National Institute of Standards and Technology",
          "@type": "org:Organization",
          "@id": "ror:05xpvk416"
        }
      ],
      "orcid": "0000-0003-0547-3117",
      "@type": "foaf:Person"
    },
    {
      "familyName": "Boyd-Graber",
      "fn": "Jordan Lee Boyd-Graber",
      "givenName": "Jordan",
      "middleName": "Lee",
      "affiliation": [
        {
          "title": "University of Maryland",
          "@type": "org:Organization"
        }
      ],
      "orcid": "0000-0002-7770-4431",
      "@type": "foaf:Person"
    }
  ],
  "annotated": "2026-07-29T12:59:02.593375",
  "revised": "2026-07-29T12:59:02.593375",
  "issued": null,
  "firstIssued": "2026-07-29T12:59:02.593375"
}