{
  "schema_version": "0.1.0",
  "record_type": "research-topic",
  "topic_id": "model-compression",
  "label": "Model Compression",
  "title": "Model Compression Research",
  "description": "Model Compression research papers in the Naser Ezzati-Jivan publication catalog.",
  "introduction": "This topic page groups Naser Ezzati-Jivan research papers related to model compression. Each linked record provides the paper's problem, method, findings, limitations, keywords, and authoritative source links.",
  "aliases": [
    "model compression"
  ],
  "search_terms": [
    "Model Compression"
  ],
  "related_topics": [],
  "canonical_url": "https://threadslab.org/research-publications/topics/model-compression.html",
  "paper_count": 1,
  "papers": [
    {
      "paper_id": "optimization-transformers-llms",
      "title": "Optimization Strategies for Enhancing Resource Efficiency in Transformers & Large Language Models",
      "year": 2025,
      "authors": [
        "Tom Wallace",
        "Beatrice M. Ombuki-Berman",
        "Naser Ezzati-Jivan"
      ],
      "page_url": "https://threadslab.org/research-publications/papers/optimization-transformers-llms/",
      "canonical_source_url": "https://doi.org/10.1145/3676151.3719379",
      "core_contribution": "The paper compares compression and optimization strategies for reducing the resource cost of Transformer and large-language-model workloads while retaining useful accuracy.",
      "tags": [
        "llm-efficiency",
        "energy-efficiency",
        "model-compression",
        "performance-engineering"
      ],
      "keywords": [
        "transformers",
        "quantization",
        "knowledge distillation",
        "pruning",
        "4-bit quantization",
        "Minitron",
        "sustainable AI"
      ]
    }
  ]
}
