{
  "schema_version": "0.6.0",
  "paper_id": "developing-a-taxonomy-for-advanced-log-parsing-techniques",
  "page_url": "https://naser.github.io/research-publications/papers/developing-a-taxonomy-for-advanced-log-parsing-techniques/",
  "title": "Developing a Taxonomy for Advanced Log Parsing Techniques",
  "title_variants": [],
  "authors": [
    "Issam Sedki",
    "Abdelwahab Hamou-Lhadj",
    "Otmane Ait Mohamed",
    "Naser Ezzati-Jivan"
  ],
  "author_details": [
    {
      "name": "Issam Sedki",
      "orcid": null,
      "profile_url": "https://dblp.org/pid/305/0448.html"
    },
    {
      "name": "Abdelwahab Hamou-Lhadj",
      "orcid": "https://orcid.org/0000-0002-3319-5006",
      "profile_url": "https://dblp.org/pid/70/2136.html"
    },
    {
      "name": "Otmane Ait Mohamed",
      "orcid": null,
      "profile_url": "https://dblp.org/pid/95/4076.html"
    },
    {
      "name": "Naser Ezzati-Jivan",
      "orcid": "https://orcid.org/0000-0003-1435-6297",
      "profile_url": "https://naser.github.io/"
    }
  ],
  "publication": {
    "year": 2025,
    "venue": "IEEE International Conference on Program Comprehension (ICPC)",
    "type": "conference paper",
    "publication_date": "2025-04-27",
    "online_date": null,
    "print_date": "2025-04-27",
    "volume": null,
    "issue": null,
    "pages": "01-12",
    "article_number": null,
    "publisher": "IEEE",
    "issn": [],
    "isbn": [],
    "crossref_type": "proceedings-article"
  },
  "publication_type": "conference paper",
  "status": "published_with_public_full_text",
  "canonical_source_url": "https://doi.org/10.1109/ICPC66645.2025.00061",
  "source_record_id": "developing-a-taxonomy-for-advanced-log-parsing-techniques-d5e39eb0f3",
  "identifiers": {
    "doi": "10.1109/ICPC66645.2025.00061"
  },
  "abstract": null,
  "abstract_source": "Author-hosted full-text PDF reviewed; abstract paraphrased for this catalog.",
  "abstract_available": false,
  "scholar_eligibility": {
    "eligible": false,
    "basis": "not-eligible",
    "note": "The page is a discovery record; it does not claim Google Scholar article-host eligibility."
  },
  "description": "The paper introduces a taxonomy of log-event characteristics that explains why different log parsers fail across systems and parser families.",
  "evidence_level": "full-text-reviewed",
  "evidence": {
    "source_basis": "full-text-reviewed",
    "coverage": "material paper sections",
    "summary_origin": "AI-assisted catalog editorial summary",
    "review_status": "catalog-reviewed; paper-author approval pending",
    "verified_on": "2026-08-09",
    "sources": [
      {
        "note": "ICPC taxonomy PDF: 16 LogHub datasets, 32,000 labelled events, eight parsers, 30 characteristics, and three taxonomy categories"
      },
      {
        "note": "ICPC taxonomy PDF: parser error counts, difficult token patterns, chi-square/effect-size results, and IPv6 exception"
      },
      {
        "note": "ICPC taxonomy PDF: limitations and hybrid/adaptive-parser future directions"
      },
      {
        "note": "Local PDF hash verified in pdf-evidence/extraction-manifest.json"
      }
    ]
  },
  "summary": {
    "core_contribution": "The paper introduces a taxonomy of log-event characteristics that explains why different log parsers fail across systems and parser families.",
    "problem": "Heterogeneous and evolving logs make template extraction unreliable; existing parser evaluations often emphasize algorithm design without identifying the log-event characteristics that cause errors across datasets and parser families.",
    "method": "The study samples 16 LogHub datasets with 2,000 manually parsed events per dataset, yielding 32,000 labeled events. Eight parsers-Drain, IPLoM, AEL, Spell, LenMa, LogMine, SHISO, and ULP-are compared against ground-truth templates. Open coding, regular expressions, named-entity recognition, and manual review identify 30 log-event characteristics (LECs) grouped into Log Event Presentation, Data Types, and Structural Arrangement of Tokens. A chi-square test of independence examines association with parsing errors.",
    "findings": "Total parser error counts are Drain 851, IPLoM 882, LenMa 883, AEL 914, ULP 923, Spell 959, SHISO 963, and LogMine 965; IPLoM timed out on Android. Unseparated token sequences and alphanumeric/special-character mixtures are repeatedly difficult. Dataset mismatch totals include Linux 93.81%, OpenStack 83.13%, BGL 73.33%, HDFS 9.85%, and Apache 38.80% in the reported table.",
    "limitations": "The corpus contains 16 public datasets and eight parsers, not industrial/proprietary logs; parser versions, host/runtime configuration, and independent-run protocol are unknown. Open coding and automated LEC detection may miss characteristics in complex datasets.",
    "future_work": "Develop hybrid parsers that combine domain knowledge with data-driven adaptation, handle difficult LECs dynamically, and promote standardized logging formats."
  },
  "tags": [
    "observability",
    "trace-analysis",
    "anomaly-detection",
    "performance-analysis"
  ],
  "keywords": [
    "log parsing",
    "log event characteristics",
    "LEC taxonomy",
    "LogHub",
    "Drain",
    "IPLoM",
    "AEL",
    "Spell",
    "LenMa",
    "LogMine",
    "SHISO",
    "ULP",
    "open coding",
    "regex",
    "NER",
    "chi-square",
    "parser errors",
    "token structure"
  ],
  "versions": [
    {
      "id": "published-version",
      "label": "Published version",
      "relation": "version-of-record",
      "title": "Developing a Taxonomy for Advanced Log Parsing Techniques",
      "url": "https://doi.org/10.1109/ICPC66645.2025.00061",
      "pdf_url": null,
      "status": "published",
      "canonical_for_citation": true
    },
    {
      "id": "public-full-text",
      "label": "Public full text",
      "relation": "source-record",
      "title": "Developing a Taxonomy for Advanced Log Parsing Techniques",
      "url": "https://users.encs.concordia.ca/~abdelw/papers/ICPC2025_LECs.pdf",
      "pdf_url": "https://users.encs.concordia.ca/~abdelw/papers/ICPC2025_LECs.pdf",
      "status": "public_full_text",
      "canonical_for_citation": false
    }
  ],
  "access": {
    "status": "published_with_public_full_text",
    "note": "The DOI is the canonical citation target; the public full-text or author-manuscript link is external and the PDF is not redistributed here.",
    "license": null
  },
  "resources": {
    "code": null,
    "data": null,
    "slides": null,
    "demo": null
  },
  "citation_guidance": {
    "when_to_cite": "Cite this paper when your work uses or compares a 30-characteristic log-event taxonomy organized into presentation, data-type, and structural-token dimensions.",
    "points": [
      "a 30-characteristic log-event taxonomy organized into presentation, data-type, and structural-token dimensions.",
      "the eight-parser/16-LogHub benchmark and its exact-template error comparison.",
      "the finding that unseparated token sequences and alphanumeric/special-character mixtures are cross-parser failure hotspots.",
      "the Linux/HDFS/Apache mismatch contrast when motivating dataset-dependent parser evaluation."
    ],
    "canonical_version_id": "published-version"
  },
  "provenance": {
    "metadata_verified_on": "2026-08-09",
    "metadata_source": [
      "ICPC taxonomy PDF: 16 LogHub datasets, 32,000 labelled events, eight parsers, 30 characteristics, and three taxonomy categories",
      "ICPC taxonomy PDF: parser error counts, difficult token patterns, chi-square/effect-size results, and IPv6 exception",
      "ICPC taxonomy PDF: limitations and hybrid/adaptive-parser future directions",
      "Local PDF hash verified in pdf-evidence/extraction-manifest.json"
    ],
    "summary_written_by": "AI-assisted",
    "summary_verified_by": "full-text-grounded catalog review; author approval pending",
    "linked_preprint_record": null,
    "author_order_note": null
  },
  "batch": {
    "phase": 2,
    "batch_label": "expanded forty-paper release",
    "status": "included_in_expanded_catalog",
    "selected_at": "2026-08-09"
  }
}
