Files
analise-artigo/analise_artigo/bib.py
Jário José 550440f4a7 Projeto analise-artigo: análise ABNT de artigos LaTeX
- texparse.py: extração de estrutura abntex2 (título, autores, resumo,
  palavras-chave, seções, citações, figuras), sem dependências externas
- bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos
- analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch
  adaptado pt-BR e heurísticas de prosa
- report.py: relatório Markdown + JSON
- artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
2026-09-25 13:24:56 -03:00

168 lines
4.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Parser simples de arquivos .bib (BibTeX).
Usa apenas a biblioteca padrão; suporta campos com valores em chaves
aninhadas e valores multi-linha.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from pathlib import Path
@dataclass
class Reference:
key: str
type: str # article, inproceedings, misc, ...
fields: dict[str, str] = field(default_factory=dict)
@property
def authors(self) -> str:
value = self.fields.get("author", "").strip()
# "X, A. and B, C. and others" -> "X, A., B, C. et al." (apenas campos author)
value = re.sub(r"\s+and\s+others\b", " et al.", value, flags=re.I)
return re.sub(r"\s+and\s+", ", ", value, flags=re.I)
@property
def title(self) -> str:
return self.fields.get("title", "").strip()
@property
def year(self) -> str:
return self.fields.get("year", "").strip()
@property
def journal(self) -> str:
return self.fields.get(
"journal", self.fields.get("booktitle", "")
).strip()
def label(self) -> str:
"""Rótulo curto autor (ano)."""
first = re.split(r",|and", self.authors)[0].strip()
if "," in first:
surname = first.split(",")[0].strip()
else:
surname = last_token(first)
return f"{surname} ({self.year})" if self.year else surname
def last_token(name: str) -> str:
parts = name.replace("-", " ").split()
return parts[-1] if parts else name
def _split_entries(text: str) -> list[tuple[str, str, str]]:
"""Retorna [(type, key, body), ...] respeitando aninhamento de chaves."""
entries: list[tuple[str, str, str]] = []
i = 0
while True:
m = re.compile(r"@(\w+)\s*\{").search(text, i)
if not m:
break
etype, key = m.group(1), ""
start = m.end()
depth = 1
j = start
while j < len(text) and depth:
if text[j] == "{":
depth += 1
elif text[j] == "}":
depth -= 1
j += 1
body = text[start : j - 1]
# chave = conteúdo até a primeira vírgula fora de chaves
depth = 0
k = 0
while k < len(body):
c = body[k]
if c == "{":
depth += 1
elif c == "}":
depth -= 1
elif c == "," and depth == 0:
break
if c == "\n" and depth == 0:
break
k += 1
key = body[:k].strip()
entries.append((etype, key, body[k + 1 :]))
i = j
return entries
def _parse_fields(body: str) -> dict[str, str]:
fields: dict[str, str] = {}
i = 0
n = len(body)
while i < n:
while i < n and body[i] in " \t\n,":
i += 1
if i >= n:
break
m = re.compile(r"(\w+)\s*=").match(body, i)
if not m:
break
fname = m.group(1).lower()
i = m.end()
while i < n and body[i] in " \t\n":
i += 1
if i >= n:
break
if body[i] == "{":
depth = 1
j = i + 1
while j < n and depth:
if body[j] == "{":
depth += 1
elif body[j] == "}":
depth -= 1
j += 1
value = body[i + 1 : j - 1]
i = j
else:
m2 = re.compile(r"([^,]+)").match(body, i)
if not m2:
i += 1
continue
value = m2.group(1)
i = m2.end()
value = value.strip()
value = re.sub(r"\s*\n\s*", " ", value)
value = value.replace("{", "").replace("}", "").strip()
value = _clean_bib_value(value)
fields[fname] = value
return fields
def _clean_bib_value(value: str) -> str:
"""Remove escapes LaTeX correntes em .bib (diacríticos, \\& etc.).
Ex.: ``Ki\\vs\\vs`` -> ``Kiss``, ``L'opez`` -> ``Lopez``,
``Tirke\\cs`` -> ``Tirkes``.
"""
value = re.sub(
r"\\[\'\"~vc](?:\{([A-Za-z])\}|([A-Za-z]))",
lambda m: m.group(1) or m.group(2),
value,
flags=re.I,
)
for esc, ch in (("\\&", "&"), ("\\$", "$"), ("\\%", "%"),
("\\_", "_"), ("\\#", "#")):
value = value.replace(esc, ch)
value = value.replace("---", " — ").replace("--", " – ")
return value
def parse_bib(bib_path: Path | str) -> dict[str, Reference]:
text = Path(bib_path).read_text(encoding="utf-8")
refs: dict[str, Reference] = {}
for etype, key, body in _split_entries(text):
if not key:
continue
refs[key] = Reference(key=key, type=etype, fields=_parse_fields(body))
return refs