<?xml version="1.0" encoding="UTF-8"?>
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:title>Fast Text Tokenization</dc:title>
  <dc:title>R package tok version 0.2.3</dc:title>
  <dc:description>
  Interfaces with the 'Hugging Face' tokenizers library to provide implementations
  of today's most used tokenizers such as the 'Byte-Pair Encoding' algorithm 
  &lt;https://huggingface.co/docs/tokenizers/index&gt;. It's extremely fast for both 
  training new vocabularies and tokenizing texts.</dc:description>
  <dc:type>Software</dc:type>
  <dc:relation>Depends: R (&gt;= 4.2.0)</dc:relation>
  <dc:relation>Imports: R6, cli</dc:relation>
  <dc:relation>Suggests: rmarkdown, testthat (&gt;= 3.0.0), hfhub (&gt;= 0.1.1), withr</dc:relation>
  <dc:creator>Tomasz Kalinowski &lt;tomasz@posit.co&gt;</dc:creator>
  <dc:publisher>Comprehensive R Archive Network (CRAN)</dc:publisher>
  <dc:contributor>Tomasz Kalinowski [ctb, cre],
  Daniel Falbel [aut],
  Regouby Christophe [ctb],
  Posit [cph]</dc:contributor>
  <dc:rights>MIT + file LICENSE (https://CRAN.R-project.org/package=tok/LICENSE)</dc:rights>
  <dc:date>2026-06-21</dc:date>
  <dc:format>application/tgz</dc:format>
  <dc:identifier>https://CRAN.R-project.org/package=tok</dc:identifier>
  <dc:identifier>doi:10.32614/CRAN.package.tok</dc:identifier>
</oai_dc:dc>
