{
 "product": "YFarmX Model Identity",
 "method": "tokenizer fingerprinting v1-50",
 "updated": "2026-08-22",
 "record_url": "https://yfarmx.com/ai/models/deepseek/deepseek-r1-distill-llama-70b/",
 "citation": "YFarmX Model Identity: DeepSeek: R1 Distill Llama 70B (deepseek/deepseek-r1-distill-llama-70b) evidence record. Measured 2026-08-21. https://yfarmx.com/ai/models/deepseek/deepseek-r1-distill-llama-70b/",
 "record": {
  "id": "deepseek/deepseek-r1-distill-llama-70b",
  "slug": "deepseek/deepseek-r1-distill-llama-70b",
  "name": "DeepSeek: R1 Distill Llama 70B",
  "lab": "DeepSeek",
  "alias": false,
  "variant_of": null,
  "created": 1737663169,
  "new_since_20_aug": false,
  "description": "DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...",
  "declared": {
   "tokenizer": "Llama3",
   "modality": "text->text",
   "input": [
    "text"
   ],
   "context": 8192,
   "max_output": 8192,
   "pricing": {
    "prompt": "0.0000008",
    "completion": "0.0000008"
   },
   "params": [
    "frequency_penalty",
    "include_reasoning",
    "max_tokens",
    "presence_penalty",
    "reasoning",
    "repetition_penalty",
    "seed",
    "stop",
    "temperature",
    "top_k",
    "top_p"
   ],
   "defaults": {},
   "reasoning": {
    "mandatory": false
   },
   "hugging_face": "deepseek-ai/DeepSeek-R1-Distill-Llama-70B"
  },
  "measurement": {
   "status": "measured",
   "at": "2026-08-21",
   "run": "catalogue-sweep",
   "overhead": 3,
   "providers": [
    "Novita"
   ],
   "clean_rows": 50,
   "corrupt_rows": 0,
   "signature": "tk_869dc1cf",
   "counts": {
    "en-prose": 14,
    "en-long": 16,
    "spaces-20": 3,
    "spaces-60": 3,
    "tabs-20": 4,
    "newlines-20": 4,
    "mixed-ws": 7,
    "digits-9": 3,
    "digits-12": 4,
    "digits-30": 10,
    "digits-sep": 7,
    "float-long": 9,
    "zh-common": 9,
    "zh-long": 10,
    "zh-rare": 18,
    "ja-kana": 9,
    "ja-kanji": 11,
    "ko": 13,
    "ru": 13,
    "ar": 10,
    "he": 14,
    "hi": 18,
    "th": 11,
    "el": 16,
    "emoji-basic": 10,
    "emoji-skin": 14,
    "emoji-zwj-family": 11,
    "emoji-zwj-x3": 33,
    "emoji-flags": 16,
    "emoji-prof": 19,
    "math": 22,
    "boxdraw": 27,
    "combining": 10,
    "cjk-ext-b": 16,
    "surrogates": 33,
    "zalgo": 36,
    "rtl-mix": 6,
    "py-code": 25,
    "py-indent": 15,
    "json": 20,
    "html": 16,
    "regex": 47,
    "camel": 7,
    "snake": 8,
    "rare-word-x5": 25,
    "repeat-tok": 40,
    "base64": 34,
    "hex": 12,
    "url": 17,
    "uuid": 27
   },
   "corrupt": {},
   "observations": [
    {
     "at": "2026-08-21",
     "run": "catalogue-sweep",
     "clean": 50
    }
   ]
  },
  "note": {
   "date": "2026-08-21",
   "text": "The mismatch was traced to the serving layer, not the label. The model's own repository tokenizer, run over the same 50 strings offline, is Llama 3's exactly (50 of 50, vocabulary 128,256), so the catalogue's Llama3 tag is right about the weights. The endpoint's reported token counts tell a different story: they match DeepSeek-V3's repository tokenizer on every string once the template boundary is folded, and the model's own tokenizer on only 18 of 50. The token accounting behind this endpoint uses a DeepSeek vocabulary the model does not read with, which sets what users are billed. Whether that is a billing-side tokenizer configuration or something else about the serving path cannot be established from token counts alone.",
   "sources": [
    {
     "label": "DeepSeek-R1-Distill-Llama-70B repository (config.json: vocab_size 128256, LlamaForCausalLM)",
     "url": "https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Llama-70B/blob/main/config.json"
    },
    {
     "label": "DeepSeek-V3 repository (config.json: vocab_size 129280)",
     "url": "https://huggingface.co/deepseek-ai/DeepSeek-V3/blob/main/config.json"
    }
   ]
  },
  "api_signature": "api_28524539",
  "api_matches": [],
  "identity": {
   "kind": "lineage",
   "family": "DeepSeek",
   "confidence": "high",
   "confidence_reason": "an exact fingerprint shared with 1 model that declares the same family, without a second signal of a different kind yet",
   "basis": [
    "exact tokenizer signature shared with 1 model"
   ],
   "contradiction": null,
   "declared_vs_measured": "partial"
  },
  "comparator": "deepseek/deepseek-chat-v3.1",
  "neighbours": [
   {
    "id": "deepseek/deepseek-chat-v3.1",
    "same": 50,
    "total": 50
   },
   {
    "id": "deepseek/deepseek-v4-pro-0813",
    "same": 48,
    "total": 50
   },
   {
    "id": "deepseek/deepseek-v4-flash-0731",
    "same": 48,
    "total": 50
   },
   {
    "id": "deepseek/deepseek-v4-flash",
    "same": 48,
    "total": 50
   },
   {
    "id": "stepfun/step-3.5-flash",
    "same": 48,
    "total": 50
   },
   {
    "id": "stepfun/step-3.7-flash",
    "same": 48,
    "total": 50
   }
  ]
 }
}