{
  "slug": "context-compression-is-not-one-thing",
  "title": "Context Compression Is Not One Thing: Readable Symbolic Re-expression vs. Coherent Summary at Matched Budget",
  "short": "CCNOT",
  "line": "telegraph",
  "line_name": "Telegraph English",
  "part": null,
  "status": "ACL ARR 2026 submission",
  "date": "2026-05-25",
  "authors": [
    "Sisong Bei",
    "Mikhail L Arbuzov",
    "Ziwei Dong",
    "Alexey Shvets",
    "Dmitri Kalaev"
  ],
  "abstract": "We study context compression for multi-hop question answering with small language models. We propose Telegraph English, a readable symbolic format that rewrites retrieved passages into structured entity-relation statements, preserving reasoning evidence at lower token cost. In controlled experiments on MuSiQue, 2Wiki, and HotpotQA, Telegraph English outperforms three matched-budget compression baselines (character-level deletion, truncation, and random subsampling) on every dataset, with gains of 13 to 20 F1 points. It also outperforms a coherent prose summary produced by the same encoder on the hardest dataset. A pre-registered depth-interaction hypothesis is null: the advantage does not grow with reasoning depth within datasets. We interpret these results as evidence that readable symbolic re-expression preserves entity content more densely than either natural language or coherent summarization at matched token budget.",
  "tldr": "Rewriting retrieved passages as pipe-separated entity-relation clauses, entities kept verbatim, beat character deletion, truncation and random subsampling at the same token budget on three multi-hop benchmarks. At a fixed budget, the form of the kept text is not a detail.",
  "pages": 13,
  "html": "https://telegrapher.ai/research/context-compression-is-not-one-thing/",
  "md": "https://telegrapher.ai/research/context-compression-is-not-one-thing.md",
  "reader": "https://telegrapher.ai/research/context-compression-is-not-one-thing/read/",
  "pdf": "https://telegrapher.ai/papers/context-compression-is-not-one-thing/context-compression-is-not-one-thing.pdf",
  "arxiv": "2606.14875",
  "openreview": null,
  "post": "https://telegrapher.ai/blog/context-compression-is-not-one-thing/",
  "bibtex": "@misc{bei2026context,\n  title         = {Context Compression Is Not One Thing: Readable Symbolic Re-expression vs. Coherent Summary at Matched Budget},\n  author        = {Bei, Sisong and Arbuzov, Mikhail L and Dong, Ziwei and Shvets, Alexey and Kalaev, Dmitri},\n  year          = {2026},\n  note          = {ACL ARR 2026 submission},\n  eprint        = {2606.14875},\n  archivePrefix = {arXiv},\n  url           = {https://telegrapher.ai/research/context-compression-is-not-one-thing/}\n}",
  "gist": [
    {
      "label": "Claim",
      "text": "At the same token budget, entity-relation rewrites beat cut-down passages by 13 to 20 F1 points"
    },
    {
      "label": "TL;DR",
      "text": "Rewriting retrieved passages as pipe-separated entity-relation clauses, entities kept verbatim, beat character deletion, truncation and random subsampling at the same token budget on three multi-hop benchmarks. At a fixed budget, the form of the kept text is not a detail."
    },
    {
      "label": "Method",
      "text": "Claude Sonnet 4.6, under one fixed prompt, rewrote the gold-plus-distractor passages for 2,417 MuSiQue, 1,500 2Wiki and 1,000 HotpotQA questions into TE, and Qwen-3.5-9B answered from TE, from the full passage, from LLMLingua-2 at rate-50, from three controls cut to TE's per-question token count, and from a same-encoder prose summary truncated to that count, scored by token-level F1 with paired-bootstrap confidence intervals and a pre-registered test of whether TE's advantage grows with hop count."
    },
    {
      "label": "Key result",
      "text": "TE lead over the three matched-budget controls: +13.6 to +20.2 F1 points; TE lead over a same-encoder prose summary at matched budget, MuSiQue: +11.94 pp; Share of TE's named entities kept by the truncated summary: 0.497"
    },
    {
      "label": "Why it matters",
      "text": "Teams that feed retrieved passages to small models under a token budget have the most riding on this."
    },
    {
      "label": "Limits",
      "text": "One encoder, Claude Sonnet 4.6 with a frozen prompt, produced every rewrite in the main experiments, so how TE depends on encoder capacity, family or prompt wording is untested."
    },
    {
      "label": "Status",
      "text": "ACL ARR 2026 submission, May 2026"
    },
    {
      "label": "Read",
      "text": "reader /research/context-compression-is-not-one-thing/read/, PDF /papers/context-compression-is-not-one-thing/context-compression-is-not-one-thing.pdf, arXiv:2606.14875 https://arxiv.org/abs/2606.14875"
    }
  ],
  "note": {
    "slug": "context-compression-is-not-one-thing",
    "claim_title": "At the same token budget, entity-relation rewrites beat cut-down passages by 13 to 20 F1 points",
    "meta_description": "Rewritten as entity-relation clauses, retrieved passages beat three cut-down controls at the same token budget on three multi-hop QA datasets.",
    "tldr": "Rewriting retrieved passages as pipe-separated entity-relation clauses, entities kept verbatim, beat character deletion, truncation and random subsampling at the same token budget on three multi-hop benchmarks. At a fixed budget, the form of the kept text is not a detail.",
    "gist": "A small model answering a multi-hop question needs the bridge entities that link one fact to the next, and squeezing a retrieved passage into a token budget tends to lose them. Telegraph English (TE) rewrites the passage instead. A frozen encoder with a fixed prompt keeps the entities verbatim and turns the connecting prose into pipe-separated clauses with short @-prefixed operators; the consumer reads the result without fine-tuning. With Qwen-3.5-9B as the consumer on MuSiQue, 2Wiki and HotpotQA, TE beat three controls cut to its per-question token count (character deletion, end-truncation, random subsampling) on every dataset, by 13.6 to 20.2 F1 points. A coherent summary from the same encoder, truncated to the same budget, lost to TE by 11.94 points on MuSiQue and could not be separated from it on the other two. Against full, uncompressed passages the picture is mixed, and HotpotQA favours the full text. A pre-registered prediction that TE's edge grows with hop count came out null.",
    "method": "Claude Sonnet 4.6, under one fixed prompt, rewrote the gold-plus-distractor passages for 2,417 MuSiQue, 1,500 2Wiki and 1,000 HotpotQA questions into TE, and Qwen-3.5-9B answered from TE, from the full passage, from LLMLingua-2 at rate-50, from three controls cut to TE's per-question token count, and from a same-encoder prose summary truncated to that count, scored by token-level F1 with paired-bootstrap confidence intervals and a pre-registered test of whether TE's advantage grows with hop count.",
    "summary_html": [
      "Given the passage <em>“Barack Obama was born in Honolulu, Hawaii. Honolulu is the capital of the state of Hawaii. Hawaii is a state in the United States.”</em>, the encoder writes <code>Barack Obama @born Honolulu | Honolulu @capital_of Hawaii | Hawaii @state_in United States.</code> All four entities come through verbatim. The connecting verbs and articles are gone, replaced by pipes and short @-prefixed operators. That re-expression is <em>Telegraph English</em> (TE). A frozen frontier model, Claude Sonnet 4.6, produces it from a fixed, task-agnostic prompt, and a small consumer, Qwen-3.5-9B, reads it in place of the passage with no fine-tuning. The test holds the budget fixed. For each question, three controls cut the original passage down to TE's token count (deleting characters, dropping the tail, sampling random tokens); separately, the same encoder writes a free-prose summary, which is then truncated to that count. Full passages and LLMLingua-2 at rate-50 serve as reference points, across MuSiQue, 2Wiki and HotpotQA.",
      "All nine comparisons with the controls favour TE, by 13.6 to 20.2 F1 points, and every paired-bootstrap 95% interval sits above zero. The truncated summary is a harder opponent. It loses to TE by 11.94 points on MuSiQue, the dataset where full passages score lowest, and cannot be separated from TE on 2Wiki or HotpotQA. The summary itself was not the problem; the cut was. On a 10-row MuSiQue sample, the summary kept roughly half of TE's named entities at TE's budget, against 78% when left whole at 1.59× the budget. Uncompressed passages split the datasets: TE leads by about five points on MuSiQue, shows no reliable difference on 2Wiki and trails by 2.34 on HotpotQA. The pre-registered prediction that TE's lead over full passages would widen with hop count did not hold, since the hop-count slopes leaned the predicted way but none was significant after false-discovery-rate correction. On a second consumer, Mistral-7B-Instruct-v0.3, the three control gaps on MuSiQue stayed positive, at a smaller size."
    ],
    "key_numbers": [
      {
        "label": "TE lead over the three matched-budget controls",
        "value": "+13.6 to +20.2 F1 points",
        "context": "character deletion, end-truncation and random subsampling on MuSiQue, 2Wiki and HotpotQA; all nine 95% intervals strictly positive"
      },
      {
        "label": "TE lead over a same-encoder prose summary at matched budget, MuSiQue",
        "value": "+11.94 pp",
        "context": "summary truncated to TE's per-question token budget; no reliable difference on 2Wiki or HotpotQA"
      },
      {
        "label": "Share of TE's named entities kept by the truncated summary",
        "value": "0.497",
        "context": "median on a 10-row MuSiQue sample at TE's budget; 0.782 before truncation, at 1.59× the budget"
      },
      {
        "label": "TE versus the full passage, HotpotQA",
        "value": "−2.34 pp",
        "context": "TE trails uncompressed text here; it leads on MuSiQue, and the interval spans zero on 2Wiki",
        "bad": true
      },
      {
        "label": "Smallest depth widening the design could rule out",
        "value": "4–5 F1 points",
        "context": "across the hop-2-to-hop-4 range at about 80% power; the pre-registered depth interaction came out null",
        "bad": true
      }
    ],
    "editorial_html": [
      "Teams that feed retrieved passages to small models under a token budget have the most riding on this. Token-scoring compressors such as LLMLingua-2 choose which tokens to keep and return a subsequence of the input; they can drop words, but they cannot rewrite the link between two entities as an operator. Holding the budget fixed isolates the form of the kept text, and form turns out to matter a great deal. The paper reads the control results this way: cutting surface text, by characters, by the tail or at random, loses the bridge entities a multi-hop reader needs. A coherent summary cut to the same length kept roughly half of TE's named entities in a ten-row check. The paper's name for the explanation is the density argument: pipe-separated clauses spend tokens on entities instead of on the articles, auxiliaries and connectives that prose needs, so at matched budget the axis that counts is entities per token.",
      "Two practical properties come with the format. The consumer needs no training, unlike hidden-state methods that compress into the consumer's latent space, and the compressed context stays readable text that a person can audit. Neither is a reason to rewrite passages that already fit. Against full passages TE led on MuSiQue, showed no reliable difference on 2Wiki and lost on HotpotQA; the evidence is for compression under a budget.",
      "The name and the rewrite-instead-of-delete idea come from <em>Telegraph English</em>, which set the approach against LLMLingua-2's token deletion. This paper holds one strong encoder fixed. <em>Evaluating Relational Context Compression at Realized Token Budgets</em> puts the rewrite through a pre-registered matrix of encoders and consumers, finds that most encoders broke the registered output schema, and argues for comparing at realized token ratios. <em>What Survives Learned Symbolic Compression?</em> turns to learned translators and holds that what a code states about its source and what a consumer recovers from it should be measured separately, a distinction that bears directly on the small-encoder question left open here."
    ],
    "limitations": "One encoder, Claude Sonnet 4.6 with a frozen prompt, produced every rewrite in the main experiments, so how TE depends on encoder capacity, family or prompt wording is untested. A pilot in which a fine-tuned Qwen-3.5-0.8B beat both LLMLingua-2 and the Sonnet encoder used oracle bridge entities at inference, ran on MuSiQue alone and sat at roughly a tenth of TE's budget; it is an upper bound on encoder substitution, not a deployable encoder. Passages are the gold-plus-distractor sets released with each benchmark rather than the output of a deployed retriever. The full-passage and summary comparisons rest on Qwen-3.5-9B. On Mistral-7B-Instruct-v0.3 only the matched-budget controls on MuSiQue give a usable comparison: long full passages did not fit at 24 GB, so the rows that did fit skew toward shorter questions, and the summary comparison was not run. The summary was compared at matched tokens, not matched entity coverage; TE and the summary were not compared at a larger shared budget; and the entity-coverage figures come from 10 MuSiQue rows. The depth null is bounded: the design rules out widenings of about 4–5 F1 points from two hops to four and cannot see smaller ones. Under exact match, TE's MuSiQue lead over full passages shrinks to +1.08 points with an interval crossing zero, consistent with part of the F1 lead being partial credit. The between-dataset pattern rests on three datasets and is reported as exploratory.",
    "concept_terms": [
      "context compression",
      "matched token budget",
      "multi-hop question answering",
      "bridge entities",
      "Telegraph English",
      "pre-registered null result"
    ],
    "references": [
      "Pan et al. (2024). LLMLingua-2: Data distillation for efficient and faithful task-agnostic prompt compression.",
      "Jiang et al. (2023). LLMLingua: Compressing prompts for accelerated inference of large language models.",
      "Xu et al. (2024). RECOMP: Improving retrieval-augmented LMs with context compression and selective augmentation.",
      "Mu et al. (2023). Learning to compress prompts with gist tokens.",
      "Trivedi et al. (2022). MuSiQue: Multihop questions via single-hop question composition."
    ],
    "source_used": "local_pdf_text",
    "source_text_ref": "C:/Users/mikea/SCRIPTS/telegrapher-site/work/paper_text/context-compression-is-not-one-thing.txt",
    "verification": {
      "claims": [
        {
          "claim": "claim_title / meta_description / tldr: at the same token budget TE beats the three cut-down controls by 13 to 20 F1 points on three multi-hop datasets",
          "verdict": "CONFIRMED",
          "source_quote": "every dataset, with gains of 13 to 20 F1 points."
        },
        {
          "claim": "TE beats character deletion, end-truncation and random subsampling on every dataset by 13.6 to 20.2 F1 points (Qwen-3.5-9B consumer)",
          "verdict": "CONFIRMED",
          "source_quote": "strictly positive and gains of +13.6 to +20.2 F1 points"
        },
        {
          "claim": "All nine control comparisons have paired-bootstrap 95% intervals above zero",
          "verdict": "CONFIRMED",
          "source_quote": "All nine intervals exclude zero from above"
        },
        {
          "claim": "Character-deletion control drops characters until TE's per-row token count is matched",
          "verdict": "CONFIRMED",
          "source_quote": "drop every k-th character with k chosen per row"
        },
        {
          "claim": "End-truncation control cuts the passage from the end to TE's token count",
          "verdict": "CONFIRMED",
          "source_quote": "The NL passage is truncated from the end at the qwen-token boundary"
        },
        {
          "claim": "Random-subsampling control samples tokens to TE's budget",
          "verdict": "CONFIRMED",
          "source_quote": "A fixed-seed uniform subsample of qwen-token positions"
        },
        {
          "claim": "Each control is sized to TE's per-question (per-row) token count",
          "verdict": "CONFIRMED",
          "source_quote": "Each control is computed against TE’s per-row qwen-token count"
        },
        {
          "claim": "A same-encoder free-prose summary is truncated to TE's per-row token count",
          "verdict": "CONFIRMED",
          "source_quote": "We run the same encoder under a freeprose summary prompt and post-truncate each summary to TE’s per-row token budget"
        },
        {
          "claim": "Holding encoder, consumer and budget fixed isolates the form of the compressed text",
          "verdict": "CONFIRMED",
          "source_quote": "isolates representation format from compression ratio"
        },
        {
          "claim": "TE beats the truncated summary by 11.94 points on MuSiQue, interval above zero",
          "verdict": "CONFIRMED",
          "source_quote": "+11.94 pp with the CI strictly positive"
        },
        {
          "claim": "TE and the truncated summary cannot be separated on 2Wiki or HotpotQA",
          "verdict": "CONFIRMED",
          "source_quote": "On 2Wiki and HotpotQA the difference is null."
        },
        {
          "claim": "MuSiQue is the dataset where full passages score lowest (replaces 'deepest questions', which Appendix I contradicts: mean hops 2.65 vs 2Wiki 3.0)",
          "verdict": "CONFIRMED",
          "source_quote": "dataset NL F1 (MuSiQue 38.0, 2Wiki 56.0, HotpotQA 69.7)"
        },
        {
          "claim": "Truncated summary keeps roughly half (median 0.497) of TE's named entities on a 10-row MuSiQue sample",
          "verdict": "CONFIRMED",
          "source_quote": "F3 retains 0.497 of TE’s named entities, versus 0.782 for F3 raw"
        },
        {
          "claim": "Untruncated summary at 1.59x TE's budget keeps 78% (0.782) of TE's named entities",
          "verdict": "CONFIRMED",
          "source_quote": "the untruncated summary at 1.59× the budget retains 78%"
        },
        {
          "claim": "The entity drop comes from the truncation, not the encoder",
          "verdict": "CONFIRMED",
          "source_quote": "is produced entirely by matched-budget truncation, not by the encoder"
        },
        {
          "claim": "TE leads full passages on MuSiQue by about five F1 points (Table 2 prints +4.75, Table 6 prints +5.24 for the same contrast; the word rounding holds under both)",
          "verdict": "CONFIRMED",
          "source_quote": "TE beats NL by +4.75 pp"
        },
        {
          "claim": "No reliable TE-vs-full difference on 2Wiki (interval spans zero)",
          "verdict": "CONFIRMED",
          "source_quote": "gap is +1.53 pp with a CI that narrowly spans zero"
        },
        {
          "claim": "TE trails full passages by 2.34 points on HotpotQA",
          "verdict": "CONFIRMED",
          "source_quote": "TE loses to NL by −2.34 pp (p < 0.01)"
        },
        {
          "claim": "Against full passages: TE leads on MuSiQue, null on 2Wiki, loses on HotpotQA",
          "verdict": "CONFIRMED",
          "source_quote": "positive significant on MuSiQue, positive null on 2Wiki, negative significant on HotpotQA"
        },
        {
          "claim": "Pre-registered depth prediction is null: slopes lean the predicted way, none significant after FDR correction",
          "verdict": "CONFIRMED",
          "source_quote": "All four within-dataset interaction slopes are direction-consistent with the prediction but none is statistically significant"
        },
        {
          "claim": "The design rules out depth widenings of about 4-5 F1 points from hop 2 to hop 4",
          "verdict": "CONFIRMED",
          "source_quote": "ruling out widenings of roughly 4–5 F1 points across the hop-2-to-hop-4 range"
        },
        {
          "claim": "...at about 80% power, and cannot distinguish smaller effects",
          "verdict": "CONFIRMED",
          "source_quote": "rules out widenings of that magnitude with ≈ 80% power but cannot distinguish smaller effects from no effect"
        },
        {
          "claim": "On Mistral-7B-Instruct-v0.3 the three control gaps on MuSiQue stay positive",
          "verdict": "CONFIRMED",
          "source_quote": "yields paired-bootstrap 95% CIs strictly positive on all three controls"
        },
        {
          "claim": "Mistral gaps are smaller than Qwen's",
          "verdict": "CONFIRMED",
          "source_quote": "at the smaller magnitude expected for a 7B consumer"
        },
        {
          "claim": "Mistral: long full passages did not fit at 24 GB; surviving full-passage rows skew to shorter questions",
          "verdict": "CONFIRMED",
          "source_quote": "long-context NL rows do not fit at 24 GB and the surviving subset is biased toward shorter (easier) questions"
        },
        {
          "claim": "Summary comparison was not run on Mistral; full-passage and summary comparisons rest on Qwen-3.5-9B",
          "verdict": "CONFIRMED",
          "source_quote": "The coherent-prose comparator was not run on Mistral for the same reason."
        },
        {
          "claim": "Encoder is Claude Sonnet 4.6 with a fixed, task-agnostic prompt",
          "verdict": "CONFIRMED",
          "source_quote": "with a fixed, task-agnostic prompt"
        },
        {
          "claim": "Encoder is frozen and the consumer needs no fine-tuning",
          "verdict": "CONFIRMED",
          "source_quote": "the encoder is frozen; the consumer needs no fine-tuning."
        },
        {
          "claim": "Encoder is a strong frontier model",
          "verdict": "CONFIRMED",
          "source_quote": "deliberately chosen as a strong frontier model"
        },
        {
          "claim": "Consumer is Qwen-3.5-9B",
          "verdict": "CONFIRMED",
          "source_quote": "The consumer is Qwen-3.5-9B"
        },
        {
          "claim": "TE keeps entities verbatim and replaces connective prose with pipe-separated clauses and short @-prefixed operators",
          "verdict": "CONFIRMED",
          "source_quote": "tissue is replaced by short @-prefixed operators"
        },
        {
          "claim": "Obama/Honolulu NL-to-TE example as quoted",
          "verdict": "CONFIRMED",
          "source_quote": "TE: “Barack Obama @born Honolulu | Honolulu @capital_of Hawaii | Hawaii @state_in United States.”"
        },
        {
          "claim": "All four entities in the example come through verbatim",
          "verdict": "CONFIRMED",
          "source_quote": "TE preserves the entity spans (“Barack Obama”, “Honolulu”, “Hawaii”, “United States”) verbatim"
        },
        {
          "claim": "Datasets and sizes: MuSiQue 2,417, 2Wiki 1,500, HotpotQA 1,000 questions",
          "verdict": "CONFIRMED",
          "source_quote": "Per-dataset n: MuSiQue 2,417; 2Wiki 1,500; HotpotQA 1,000."
        },
        {
          "claim": "Passages are the gold-plus-distractor sets released with each benchmark, not a deployed retriever's output",
          "verdict": "CONFIRMED",
          "source_quote": "contexts released with each benchmark"
        },
        {
          "claim": "LLMLingua-2 at rate-50 is a reference baseline",
          "verdict": "CONFIRMED",
          "source_quote": "LLMLingua-2 at rate-50 (Pan et al., 2024)."
        },
        {
          "claim": "Scoring is token-level F1",
          "verdict": "CONFIRMED",
          "source_quote": "token-level F1, with binary correctness"
        },
        {
          "claim": "Paired-bootstrap 95% confidence intervals",
          "verdict": "CONFIRMED",
          "source_quote": "We use paired-bootstrap 95% confidence intervals over questions"
        },
        {
          "claim": "Pre-registered test of whether TE's advantage grows with hop count",
          "verdict": "CONFIRMED",
          "source_quote": "a primary hypothesis that TE’s advantage over NL grows with the reasoning depth of the question."
        },
        {
          "claim": "Token-scoring compressors return a subsequence of the input",
          "verdict": "CONFIRMED",
          "source_quote": "the output is a subsequence of the input"
        },
        {
          "claim": "Token scoring cannot rewrite connective text into operators at matched budget",
          "verdict": "CONFIRMED",
          "source_quote": "cannot produce that reformatting at matched budget"
        },
        {
          "claim": "Hidden-state methods compress into the consumer's latent space and need consumer-side training; TE needs none and stays auditable text",
          "verdict": "CONFIRMED",
          "source_quote": "operate in the consumer’s latent space and require consumer-side training"
        },
        {
          "claim": "Paper's reading: the cut-down controls lose the bridge entities a multi-hop reader needs",
          "verdict": "CONFIRMED",
          "source_quote": "do not reexpress content lose the bridge entities a multi-hop reader needs"
        },
        {
          "claim": "The density argument: at matched budget the relevant axis is entities per token",
          "verdict": "CONFIRMED",
          "source_quote": "We name this the density argument"
        },
        {
          "claim": "One encoder and frozen prompt in the main experiments; capacity, family and prompt wording untested",
          "verdict": "CONFIRMED",
          "source_quote": "We do not vary encoder capacity, encoder family, or prompt phrasing."
        },
        {
          "claim": "Pilot: fine-tuned Qwen-3.5-0.8B beat LLMLingua-2 and the Sonnet encoder at roughly a tenth of TE's budget",
          "verdict": "CONFIRMED",
          "source_quote": "beats both LLMLingua-2 and the Sonnet TE encoder at roughly 10% of the qwen-token budget"
        },
        {
          "claim": "Pilot used oracle bridge entities, ran on MuSiQue only, and is an upper bound, not deployable",
          "verdict": "CONFIRMED",
          "source_quote": "it is an upper bound on encoder substitutability rather than a deployable alternative"
        },
        {
          "claim": "Summary compared at matched tokens, not matched entity coverage",
          "verdict": "CONFIRMED",
          "source_quote": "a comparison at matched entity coverage would allow the prose representation a larger budget"
        },
        {
          "claim": "TE and the summary were not compared at a larger shared budget",
          "verdict": "CONFIRMED",
          "source_quote": "we do not test TE at 1.59× budget against coherent prose at 1.59× budget"
        },
        {
          "claim": "Under exact match TE's MuSiQue lead over full passages is +1.08 with an interval crossing zero, consistent with partial credit",
          "verdict": "CONFIRMED",
          "source_quote": "+1.08 pp, 95% CI [−0.58, +2.69]"
        },
        {
          "claim": "Between-dataset pattern rests on three datasets and is exploratory",
          "verdict": "CONFIRMED",
          "source_quote": "report the gradient as an exploratory observation only"
        },
        {
          "claim": "Other paper (abstracts.yaml, telegraph-english): introduced TE as a rewrite set against LLMLingua-2's token deletion",
          "verdict": "CONFIRMED",
          "source_quote": "TE performs a full semantic rewrite"
        },
        {
          "claim": "Other paper (abstracts.yaml, relational-context-compression): pre-registered encoder-consumer matrix; most encoders broke the registered schema; argues for realized token ratios",
          "verdict": "CONFIRMED",
          "source_quote": "Six of seven encoders violated the registered output schema"
        },
        {
          "claim": "Other paper (abstracts.yaml, what-survives-learned-symbolic-compression): learned translators; source fidelity and consumer recovery measured separately",
          "verdict": "CONFIRMED",
          "source_quote": "should be measured separately"
        },
        {
          "claim": "Reference present: Pan et al. (2024) LLMLingua-2",
          "verdict": "CONFIRMED",
          "source_quote": "Data distillation for efficient and faithful taskagnostic prompt compression"
        },
        {
          "claim": "Reference present: Jiang et al. (2023) LLMLingua",
          "verdict": "CONFIRMED",
          "source_quote": "LLMLingua: Compressing prompts for accelerated inference of large language models"
        },
        {
          "claim": "Reference present: Xu et al. (2024) RECOMP",
          "verdict": "CONFIRMED",
          "source_quote": "COMP: Improving retrieval-augmented LMs with context compression and selective augmentation"
        },
        {
          "claim": "Reference present: Mu et al. (2023) gist tokens",
          "verdict": "CONFIRMED",
          "source_quote": "Learning to compress prompts with gist tokens"
        },
        {
          "claim": "Reference present: Trivedi et al. (2022) MuSiQue",
          "verdict": "CONFIRMED",
          "source_quote": "MuSiQue: Multihop questions via single-hop question composition"
        }
      ],
      "revised": true,
      "status": "pass"
    }
  }
}