{
  "slug": "relational-context-compression",
  "title": "Evaluating Relational Context Compression at Realized Token Budgets",
  "short": "RCC",
  "line": "telegraph",
  "line_name": "Telegraph English",
  "part": null,
  "status": "ICLR 2027 submission",
  "date": "2026-09-18",
  "authors": [
    "Sisong Bei",
    "Mikhail L Arbuzov",
    "Ziwei Dong",
    "Yan Han",
    "Dmitri Kalaev",
    "Yanxin Zhang",
    "Alexey Shvets"
  ],
  "abstract": "Context compression must preserve answerable information within a token budget, yet shorter text alone cannot isolate the effect of its representation. We study Telegraph English, which rewrites retrieved passages into relational statements, using a pre-registered encoder–consumer matrix and an open compiler trained by distillation. Six of seven encoders violated the registered output schema on most or all inputs. Only 33 of 9,600 budgeted outputs fell within the registered band: at or below the requested budget and no more than two percent short. These findings motivate evaluating answer F1 at realized token ratios. The matrix was collected, but all registered confirmatory comparisons were unavailable because their comparator or calibration conditions were missing or ineligible. The remaining answer-F1 comparisons with full prose are descriptive and score construction failures as empty answers. The second compiler failed its pre-registered acceptance test against a budget-targeted pruning baseline. Its out-of-distribution component was unscorable because answer keys were absent, and citation-reach judging was not reached. Context-compression comparisons require measured token budgets, explicit construction failures, and available comparator evidence.",
  "tldr": "Asked to rewrite passages into relational text at a set token budget, two LLM encoders landed in the registered band on 33 of 9,600 outputs, and their median output ran long. Compressors should be compared at delivered lengths.",
  "pages": 19,
  "html": "https://telegrapher.ai/research/relational-context-compression/",
  "md": "https://telegrapher.ai/research/relational-context-compression.md",
  "reader": "https://telegrapher.ai/research/relational-context-compression/read/",
  "pdf": "https://telegrapher.ai/papers/relational-context-compression/relational-context-compression.pdf",
  "arxiv": null,
  "openreview": "https://openreview.net/forum?id=jQOAf9MQpv",
  "post": "https://telegrapher.ai/blog/relational-context-compression/",
  "bibtex": "@misc{bei2026evaluating,\n  title         = {Evaluating Relational Context Compression at Realized Token Budgets},\n  author        = {Bei, Sisong and Arbuzov, Mikhail L and Dong, Ziwei and Han, Yan and Kalaev, Dmitri and Zhang, Yanxin and Shvets, Alexey},\n  year          = {2026},\n  note          = {ICLR 2027 submission},\n  url           = {https://telegrapher.ai/research/relational-context-compression/}\n}",
  "gist": [
    {
      "label": "Claim",
      "text": "Two LLM encoders hit a requested token budget on 33 of 9,600 rewrites; the median ran long"
    },
    {
      "label": "TL;DR",
      "text": "Asked to rewrite passages into relational text at a set token budget, two LLM encoders landed in the registered band on 33 of 9,600 outputs, and their median output ran long. Compressors should be compared at delivered lengths."
    },
    {
      "label": "Method",
      "text": "Five LLM encoders (GPT-5.6-sol, Sonnet 5, Llama 4 Maverick, Magistral Small, Ministral 8B) rewrote 1,200 MuSiQue, 2WikiMultiHopQA and HotpotQA contexts into TE at natural length, and two of them also at requested ratios between 0.25 and 0.70, for Qwen3.5-9B, Llama3.1-8B-Instruct and Mistral-7B-Instruct-v0.3 readers with full prose as the reference; separately, a Qwen3-8B compiler distilled from Qwen3-Next-80B plans was scored against budget-targeted LongLLMLingua over the same requested grid."
    },
    {
      "label": "Key result",
      "text": "Budgeted outputs inside the registered band: 33 of 9,600; Mean token ratio for Sonnet 5 at a requested 0.25: 0.6574; Encoders that broke the output schema on most or all inputs: Six of seven"
    },
    {
      "label": "Why it matters",
      "text": "Whoever is choosing between context compressors for a retrieval or agent pipeline has a stake here, more so when one candidate rewrites rather than deletes."
    },
    {
      "label": "Limits",
      "text": "Every registered confirmatory comparison was unavailable: the same-encoder prose rewrite failed its development requirement, the pruning baseline had no matrix reader pass, a second baseline had unresolved usage rights, and the targeted-prompt calibration was not generated."
    },
    {
      "label": "Status",
      "text": "ICLR 2027 submission, September 2026"
    },
    {
      "label": "Read",
      "text": "reader /research/relational-context-compression/read/, PDF /papers/relational-context-compression/relational-context-compression.pdf"
    }
  ],
  "note": {
    "slug": "relational-context-compression",
    "claim_title": "Two LLM encoders hit a requested token budget on 33 of 9,600 rewrites; the median ran long",
    "meta_description": "Telegraph English encoders rarely delivered the token budget they were asked for. Compression comparisons need measured lengths and counted failures.",
    "tldr": "Asked to rewrite passages into relational text at a set token budget, two LLM encoders landed in the registered band on 33 of 9,600 outputs, and their median output ran long. Compressors should be compared at delivered lengths.",
    "gist": "A requested compression ratio is easy to treat as a shared condition across methods. For a rewriting compressor it is an instruction, not a measurement: the encoder writes a new string, and how long it comes out depends on the wording. This study follows Telegraph English (TE), which rewrites retrieved passages into relational statements, from the budget request to the text a reader model receives and the answer it gives. Its pre-registered matrix crosses five encoders with three reader models over 1,200 multi-hop questions; in a wider census of seven encoders, six broke the required output schema on most or all inputs. Of 9,600 budgeted outputs, 33 landed in the registered band, and median lengths exceeded the request across the grid. A separate distilled compiler enforced its token ceiling by construction, yet scored far below a budget-targeted LongLLMLingua baseline and failed its acceptance test. The registered comparisons that would isolate relational form from prose were all unavailable, so the matrix results are descriptive.",
    "method": "Five LLM encoders (GPT-5.6-sol, Sonnet 5, Llama 4 Maverick, Magistral Small, Ministral 8B) rewrote 1,200 MuSiQue, 2WikiMultiHopQA and HotpotQA contexts into TE at natural length, and two of them also at requested ratios between 0.25 and 0.70, for Qwen3.5-9B, Llama3.1-8B-Instruct and Mistral-7B-Instruct-v0.3 readers with full prose as the reference; separately, a Qwen3-8B compiler distilled from Qwen3-Next-80B plans was scored against budget-targeted LongLLMLingua over the same requested grid.",
    "summary_html": [
      "“Orin’s adviser is Sana, and Sana works at North Lab” can become <code>Orin | adviser | Sana; Sana | workplace | North Lab</code>; the shared name joins the two relations a reader needs to say where Orin’s adviser works. That rewrite is <em>Telegraph English</em> (TE), and the study follows it from a budget request to a reader’s answer. In a pre-registered matrix, five encoders rewrote 1,200 multi-hop contexts and three reader models answered from what came back. Full prose served as the common reference. Construction failures stayed in, as empty answers. Two of the encoders were also given token targets between 0.25 and 0.70 of the full-prose length, plus one revision after being told their count; to count as on budget, an output had to sit at or below its target and no more than two percent short. A second study reached TE by another route. An open compiler, distilled from a larger teacher, writes a source-linked plan, and a fixed serializer renders the longest variant that fits the token ceiling, without deleting a proposition or padding.",
      "Most encoders stumbled at the interface, before any reader was involved. Six of seven in a wider census broke the output schema on most or all inputs; GPT-5.6-sol was the sole encoder to return every envelope correctly, and a mechanical recovery pass salvaged readable text from most of the malformed ones. Of 9,600 budgeted outputs, 33 landed in the band. Both encoders’ median lengths stayed above the request across the grid, and asked for a quarter of the context, Sonnet 5 delivered about two thirds. At natural length, which way TE moved relative to full prose depended on the reader: every TE encoder scored above full prose with the Mistral reader and below it with the Llama and Qwen readers, and the encoder ranking shifted as well. The compiler had the converse problem. Its in-domain plans were mostly valid and its text never overran the ceiling, yet its requested-budget F1 area was 0.0658 against 0.3936 for the pruning baseline, no reader family passed, and it failed its pre-registered acceptance test."
    ],
    "key_numbers": [
      {
        "label": "Budgeted outputs inside the registered band",
        "value": "33 of 9,600",
        "context": "two anchor encoders at four requested ratios; band is at or below target and no more than two percent short",
        "bad": true
      },
      {
        "label": "Mean token ratio for Sonnet 5 at a requested 0.25",
        "value": "0.6574",
        "context": "dataset-balanced context tokens over full prose, Qwen tokenizer, construction failures placed at the requested ratio; GPT-5.6-sol: 0.3496 at the same request",
        "bad": true
      },
      {
        "label": "Encoders that broke the output schema on most or all inputs",
        "value": "Six of seven",
        "context": "generation census including two extra Ministral sizes; recovered text was still available for most rows",
        "bad": true
      },
      {
        "label": "Dataset-balanced paired contrast, TE minus full prose",
        "value": "−0.0251",
        "context": "natural-length matrix, descriptive; 95% interval −0.0356 to −0.0146; construction failures scored as zero"
      },
      {
        "label": "Requested-budget F1 area of the distilled compiler",
        "value": "0.0658",
        "context": "in-domain datasets; 0.3936 for budget-targeted LongLLMLingua, difference −0.328 (95% interval −0.349 to −0.307)",
        "bad": true
      }
    ],
    "editorial_html": [
      "Whoever is choosing between context compressors for a retrieval or agent pipeline has a stake here, more so when one candidate rewrites rather than deletes. Deletion selects tokens, so a requested ratio acts directly on what is kept. A rewriter writes a new string whose length follows its wording, which makes the request an instruction. Lined up at the same requested ratio, two rewriters can hand a reader very different amounts of text; at a requested quarter, Sonnet 5’s output was nearly twice the length of GPT-5.6-sol’s. What the paper asks for instead is a record of what was delivered: the string itself, its token count under the reader’s own tokenizer, and whether the question and template were counted.",
      "Construction failures go in that record too. Dropping the rows where no usable text came back would change what is being scored, while keeping them as zero-F1 answers measures the whole procedure, including whether it delivered an input at all. The compiler shows the converse. A hard ceiling caps length and says nothing about content: the serializer held every ceiling it was given, and the answers still fell far below a pruning baseline.",
      "Earlier work in the Telegraph English line compared the representation with token-matched controls and post-truncated prose summaries. This paper turns to the procedure that produces it, from request to delivered string. The controlled question of form, relations against a same-encoder prose rewrite at the same delivered length, was registered here and could not be run. It stays open."
    ],
    "limitations": "Every registered confirmatory comparison was unavailable: the same-encoder prose rewrite failed its development requirement, the pruning baseline had no matrix reader pass, a second baseline had unresolved usage rights, and the targeted-prompt calibration was not generated. The matrix results against full prose are therefore descriptive, with an unadjusted interval and failures scored as zero. Recovering text from a malformed envelope shows that text exists, not that it preserves the source, and a valid compiler plan passes structural checks without full span or semantic validation. The compiler was compared on the requested grid, and whether the baseline delivered equal token counts there is unverified. Its out-of-distribution component could not be scored because the 500 QASPER questions had no answers; citation-reach judging was prepared but never executed; and the second compiler changed several target properties at once, which leaves the contribution of each unresolved. Where along plan, serialization and reading the evidence was lost is not located. Scores use a modified token-overlap F1 that differs from multiset F1 when words repeat, the data are English multi-hop and research-paper QA on the named model versions, and raw contexts, predictions and compiler weights are missing from the archive, so the experiments cannot be rerun from it.",
    "concept_terms": [
      "context compression",
      "realized token ratio",
      "token budget compliance",
      "Telegraph English",
      "pre-registration",
      "encoder–consumer matrix"
    ],
    "references": [
      "Arbuzov et al. (2026). Telegraph English: Semantic prompt compression via structured symbolic rewriting.",
      "Bei et al. (2026). Context compression is not one thing: Readable symbolic re-expression vs. coherent summary at matched budget.",
      "Jiang et al. (2024). LongLLMLingua: Accelerating and enhancing LLMs in long context scenarios via prompt compression.",
      "Jiang et al. (2023). LLMLingua: Compressing prompts for accelerated inference of large language models.",
      "van Miltenburg et al. (2021). Preregistering NLP research."
    ],
    "source_used": "local_pdf_text",
    "source_text_ref": "C:/Users/mikea/SCRIPTS/telegrapher-site/work/paper_text/relational-context-compression.txt",
    "verification": {
      "claims": [
        {
          "claim": "33 of 9,600 budgeted outputs fell within the registered band",
          "verdict": "CONFIRMED",
          "source_quote": "Only 33 of 9,600 budgeted outputs fell within the registered"
        },
        {
          "claim": "The band: at or below the requested budget and no more than two percent short",
          "verdict": "CONFIRMED",
          "source_quote": "at or below the requested budget and no more than two percent short"
        },
        {
          "claim": "The 9,600 outputs come from the two anchor encoders at four requested ratios (2 x 4 x 1,200)",
          "verdict": "CONFIRMED",
          "source_quote": "with r ∈ {0.25, 0.40, 0.55, 0.70}"
        },
        {
          "claim": "Both anchors' median lengths stay above the request across the grid",
          "verdict": "CONFIRMED",
          "source_quote": "Both anchors produce median lengths above their requests throughout"
        },
        {
          "claim": "Five matrix encoders: GPT-5.6-sol, Sonnet 5, Llama 4 Maverick, Magistral Small, Ministral 8B",
          "verdict": "CONFIRMED",
          "source_quote": "The encoder instances are GPT-5.6-sol, Sonnet 5, Llama 4 Maverick, Magistral Small,"
        },
        {
          "claim": "Three readers: Qwen3.5-9B, Llama3.1-8B-Instruct, Mistral-7B-Instruct-v0.3",
          "verdict": "CONFIRMED",
          "source_quote": "The consumers are Qwen3.5-9B, Llama3.1-8B-Instruct, and Mistral-7B-Instruct-v0.3."
        },
        {
          "claim": "1,200 fixed multi-hop examples from MuSiQue, 2WikiMultiHopQA and HotpotQA",
          "verdict": "CONFIRMED",
          "source_quote": "The matrix contains 1,200 fixed examples"
        },
        {
          "claim": "Budgeted targets between 0.25 and 0.70 of full prose, plus one token-count revision",
          "verdict": "CONFIRMED",
          "source_quote": "receives a target between 0.25 and 0.70 of"
        },
        {
          "claim": "Natural-length TE gets no requested ratio",
          "verdict": "CONFIRMED",
          "source_quote": "Natural-length Telegraph English receives no requested compression ratio."
        },
        {
          "claim": "Full prose is the common reference; construction failures stay in as empty answers with F1 zero",
          "verdict": "CONFIRMED",
          "source_quote": "retains construction failures as empty"
        },
        {
          "claim": "Orin/Sana worked example and its relational rewrite",
          "verdict": "CONFIRMED",
          "source_quote": "Orin | adviser | Sana; Sana | workplace | North Lab."
        },
        {
          "claim": "The pre-registered design: encoder-consumer matrix plus an open compiler trained by distillation",
          "verdict": "CONFIRMED",
          "source_quote": "using a pre-registered encoder–consumer matrix and an open compiler trained"
        },
        {
          "claim": "Six of seven encoders in the wider census broke the output schema on most or all inputs",
          "verdict": "CONFIRMED",
          "source_quote": "Six of seven encoders violate this"
        },
        {
          "claim": "The census adds two more Ministral sizes to the matrix encoders",
          "verdict": "CONFIRMED",
          "source_quote": "the smaller and larger Ministral variants"
        },
        {
          "claim": "GPT-5.6-sol is the sole encoder that returned every envelope correctly",
          "verdict": "CONFIRMED",
          "source_quote": "GPT-5.6-sol is the sole encoder that supplies"
        },
        {
          "claim": "Mechanical recovery salvaged readable text from most malformed envelopes; usable text exists for most rows (Table 3: 1,113 to 1,200 of 1,200 per encoder) [corrected from 'the malformed ones']",
          "verdict": "CONFIRMED",
          "source_quote": "A mechanical recovery pass extracts readable bodies from some malformed envelopes."
        },
        {
          "claim": "Recovered text shows availability, not preservation of the source",
          "verdict": "CONFIRMED",
          "source_quote": "recovery checks availability rather than semantic fidelity"
        },
        {
          "claim": "Sonnet 5 at a requested 0.25: mean token ratio 0.6574 under the Qwen tokenizer ('about two thirds'), failures placed at the requested ratio (Table 5 cell) [label corrected from 'mean realized ratio']",
          "verdict": "CONFIRMED",
          "source_quote": "uses measured lengths for successes and requested positions for construction failures"
        },
        {
          "claim": "GPT-5.6-sol at a requested 0.25: mean token ratio 0.3496 under the Qwen tokenizer (Table 5 cell)",
          "verdict": "CONFIRMED",
          "source_quote": "Table 5: Dataset-balanced frontier points by encoder, consumer, and requested ratio."
        },
        {
          "claim": "Derived: at a requested quarter Sonnet 5's output was nearly twice GPT-5.6-sol's (0.6574/0.3496 = 1.88 Qwen; 0.6512/0.3338 = 1.95 Llama; 0.6708/0.3566 = 1.88 Mistral)",
          "verdict": "CONFIRMED",
          "source_quote": "Two encoders given the same requested budget can therefore present different amounts of context"
        },
        {
          "claim": "Every TE encoder scored above full prose with the Mistral reader and below it with Llama and Qwen (rechecked against all 15 Table 1 cells)",
          "verdict": "CONFIRMED",
          "source_quote": "Every relational-text entry in the Mistral"
        },
        {
          "claim": "The encoder ranking changes with the reader",
          "verdict": "CONFIRMED",
          "source_quote": "The encoder ordering also changes with the consumer."
        },
        {
          "claim": "Dataset-balanced paired contrast TE minus full prose = −0.0251, 95% interval −0.0356 to −0.0146",
          "verdict": "CONFIRMED",
          "source_quote": "The dataset-balanced paired contrast is −0.0251, with a 95% interval from −0.0356 to −0.0146."
        },
        {
          "claim": "The contrast is natural-length (45 cells = 3 datasets x 5 encoders x 3 consumers), descriptive and unadjusted",
          "verdict": "CONFIRMED",
          "source_quote": "The descriptive interval is unadjusted"
        },
        {
          "claim": "Matrix results against full prose are descriptive",
          "verdict": "CONFIRMED",
          "source_quote": "The remaining answer-F1 comparisons with full prose are descriptive"
        },
        {
          "claim": "Compiler: Qwen3-8B student distilled from Qwen3-Next-80B teacher plans",
          "verdict": "CONFIRMED",
          "source_quote": "The open teacher is Qwen3-Next-80B and the student is Qwen3-8B."
        },
        {
          "claim": "The compiler writes a source-linked plan rendered by a fixed serializer",
          "verdict": "CONFIRMED",
          "source_quote": "model emits a source-linked plan, and a fixed serializer supplies the text for reading"
        },
        {
          "claim": "Serializer renders the longest fitting variant, neither deleting propositions nor padding",
          "verdict": "CONFIRMED",
          "source_quote": "It neither deletes propositions nor pads a short representation to meet the target."
        },
        {
          "claim": "Compiler text never overran its ceiling (enforced by construction)",
          "verdict": "CONFIRMED",
          "source_quote": "Its fixed serializer enforces a token"
        },
        {
          "claim": "A ceiling caps length without securing content; budget can go unused [corrected from 'guarantees length']",
          "verdict": "CONFIRMED",
          "source_quote": "satisfy a ceiling while leaving substantial budget unused."
        },
        {
          "claim": "In-domain plans were mostly valid",
          "verdict": "CONFIRMED",
          "source_quote": "The in-domain sources yield mostly valid plans"
        },
        {
          "claim": "Requested-budget F1 area: compiler 0.0658 vs budget-targeted LongLLMLingua 0.3936, in-domain",
          "verdict": "CONFIRMED",
          "source_quote": "The width-normalized area is 0.0658 for the compiler and"
        },
        {
          "claim": "Area difference −0.328, 95% interval −0.349 to −0.307",
          "verdict": "CONFIRMED",
          "source_quote": "The difference is −0.328, with a 95% interval"
        },
        {
          "claim": "No reader family met the positive-contrast criterion",
          "verdict": "CONFIRMED",
          "source_quote": "None of the three consumer families meets the required positive-contrast"
        },
        {
          "claim": "The compiler failed its pre-registered acceptance test",
          "verdict": "CONFIRMED",
          "source_quote": "The second compiler failed its pre-registered acceptance"
        },
        {
          "claim": "Compiler compared on the same requested grid; equal delivered token counts for the baseline unverified",
          "verdict": "CONFIRMED",
          "source_quote": "with realized-budget equivalence unverified in the final"
        },
        {
          "claim": "Every registered confirmatory comparison unavailable: direct-prose development failure, no matrix reader pass for the pruning baseline, unresolved usage rights, targeted-prompt calibration absent",
          "verdict": "CONFIRMED",
          "source_quote": "Its development requirement failed; the pruning baseline had no matrix consumer"
        },
        {
          "claim": "The same-encoder prose rewrite at the same delivered length was the registered form control and could not be run",
          "verdict": "CONFIRMED",
          "source_quote": "Its development failure prevents this paired representation comparison."
        },
        {
          "claim": "Keeping construction failures as zero measures the whole procedure, including delivery of an input",
          "verdict": "CONFIRMED",
          "source_quote": "Keeping construction failures in the scored denominator evaluates the complete procedure"
        },
        {
          "claim": "A requested rate is not a shared condition across methods; a rewriter's length depends on its wording",
          "verdict": "CONFIRMED",
          "source_quote": "whereas rewriting generates a new string whose length depends on its wording"
        },
        {
          "claim": "Report the delivered string, its consumer-tokenizer count and prompt overhead",
          "verdict": "CONFIRMED",
          "source_quote": "inclusion of prompt overhead also belong with the reported ratio."
        },
        {
          "claim": "Compressors should be compared at delivered (realized) lengths",
          "verdict": "CONFIRMED",
          "source_quote": "These findings motivate evaluating answer F1 at realized token ratios."
        },
        {
          "claim": "Earlier TE work compared the form with token-matched controls and post-truncated prose summaries [rephrased; 'share one representation' removed]",
          "verdict": "CONFIRMED",
          "source_quote": "subsequent work compares this form with token-matched controls and post-truncated"
        },
        {
          "claim": "A valid compiler plan passes structural checks, not full span or semantic validation",
          "verdict": "CONFIRMED",
          "source_quote": "Its valid-plan count does not check every span boundary or"
        },
        {
          "claim": "OOD component unscorable: the 500 QASPER questions had no answers",
          "verdict": "CONFIRMED",
          "source_quote": "The terminal answer file has no answers for the 500 QASPER"
        },
        {
          "claim": "Citation-reach judging prepared but never executed",
          "verdict": "CONFIRMED",
          "source_quote": "Citation-reach judging was prepared but never executed."
        },
        {
          "claim": "Second compiler changed several target properties together; contribution of each unresolved",
          "verdict": "CONFIRMED",
          "source_quote": "These joint changes leave the contribution of each target property unresolved."
        },
        {
          "claim": "Where the evidence was lost along plan, serialization and reading is not located",
          "verdict": "CONFIRMED",
          "source_quote": "Locating the lost evidence within that path requires the generated plan"
        },
        {
          "claim": "Modified token-overlap F1 differs from multiset F1 when words repeat",
          "verdict": "CONFIRMED",
          "source_quote": "It therefore differs from ordinary multiset token F1 when a word repeats."
        },
        {
          "claim": "English multi-hop and research-paper QA on the named model versions",
          "verdict": "CONFIRMED",
          "source_quote": "Across the fixed English question-answering data and tested model versions"
        },
        {
          "claim": "Raw contexts, predictions and compiler weights missing; experiments cannot be rerun from the archive",
          "verdict": "CONFIRMED",
          "source_quote": "rerun the complete experiments from this archive alone."
        },
        {
          "claim": "Reference: Arbuzov et al. (2026), Telegraph English",
          "verdict": "CONFIRMED",
          "source_quote": "English: Semantic prompt compression via structured symbolic rewriting, 2026."
        },
        {
          "claim": "Reference: Bei et al. (2026), Context compression is not one thing",
          "verdict": "CONFIRMED",
          "source_quote": "is not one thing: Readable symbolic re-expression vs. coherent summary at matched budget, 2026."
        },
        {
          "claim": "Reference: Jiang et al. (2024), LongLLMLingua",
          "verdict": "CONFIRMED",
          "source_quote": "LongLLMLingua: Accelerating and enhancing LLMs in long context scenarios via prompt"
        },
        {
          "claim": "Reference: Jiang et al. (2023), LLMLingua",
          "verdict": "CONFIRMED",
          "source_quote": "LLMLingua: Compressing prompts for accelerated inference of large language models."
        },
        {
          "claim": "Reference: van Miltenburg et al. (2021), Preregistering NLP research",
          "verdict": "CONFIRMED",
          "source_quote": "Preregistering NLP research."
        }
      ],
      "revised": true,
      "status": "pass"
    }
  }
}