- texparse.py: extração de estrutura abntex2 (título, autores, resumo, palavras-chave, seções, citações, figuras), sem dependências externas - bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos - analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch adaptado pt-BR e heurísticas de prosa - report.py: relatório Markdown + JSON - artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
168 lines
4.7 KiB
Python
168 lines
4.7 KiB
Python
"""Parser simples de arquivos .bib (BibTeX).
|
||
|
||
Usa apenas a biblioteca padrão; suporta campos com valores em chaves
|
||
aninhadas e valores multi-linha.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
from dataclasses import dataclass, field
|
||
from pathlib import Path
|
||
|
||
|
||
@dataclass
|
||
class Reference:
|
||
key: str
|
||
type: str # article, inproceedings, misc, ...
|
||
fields: dict[str, str] = field(default_factory=dict)
|
||
|
||
@property
|
||
def authors(self) -> str:
|
||
value = self.fields.get("author", "").strip()
|
||
# "X, A. and B, C. and others" -> "X, A., B, C. et al." (apenas campos author)
|
||
value = re.sub(r"\s+and\s+others\b", " et al.", value, flags=re.I)
|
||
return re.sub(r"\s+and\s+", ", ", value, flags=re.I)
|
||
|
||
@property
|
||
def title(self) -> str:
|
||
return self.fields.get("title", "").strip()
|
||
|
||
@property
|
||
def year(self) -> str:
|
||
return self.fields.get("year", "").strip()
|
||
|
||
|
||
@property
|
||
def journal(self) -> str:
|
||
return self.fields.get(
|
||
"journal", self.fields.get("booktitle", "")
|
||
).strip()
|
||
|
||
def label(self) -> str:
|
||
"""Rótulo curto autor (ano)."""
|
||
first = re.split(r",|and", self.authors)[0].strip()
|
||
if "," in first:
|
||
surname = first.split(",")[0].strip()
|
||
else:
|
||
surname = last_token(first)
|
||
return f"{surname} ({self.year})" if self.year else surname
|
||
|
||
|
||
def last_token(name: str) -> str:
|
||
parts = name.replace("-", " ").split()
|
||
return parts[-1] if parts else name
|
||
|
||
|
||
def _split_entries(text: str) -> list[tuple[str, str, str]]:
|
||
"""Retorna [(type, key, body), ...] respeitando aninhamento de chaves."""
|
||
entries: list[tuple[str, str, str]] = []
|
||
i = 0
|
||
while True:
|
||
m = re.compile(r"@(\w+)\s*\{").search(text, i)
|
||
if not m:
|
||
break
|
||
etype, key = m.group(1), ""
|
||
start = m.end()
|
||
depth = 1
|
||
j = start
|
||
while j < len(text) and depth:
|
||
if text[j] == "{":
|
||
depth += 1
|
||
elif text[j] == "}":
|
||
depth -= 1
|
||
j += 1
|
||
body = text[start : j - 1]
|
||
|
||
# chave = conteúdo até a primeira vírgula fora de chaves
|
||
depth = 0
|
||
k = 0
|
||
while k < len(body):
|
||
c = body[k]
|
||
if c == "{":
|
||
depth += 1
|
||
elif c == "}":
|
||
depth -= 1
|
||
elif c == "," and depth == 0:
|
||
break
|
||
if c == "\n" and depth == 0:
|
||
break
|
||
k += 1
|
||
key = body[:k].strip()
|
||
entries.append((etype, key, body[k + 1 :]))
|
||
i = j
|
||
return entries
|
||
|
||
|
||
def _parse_fields(body: str) -> dict[str, str]:
|
||
fields: dict[str, str] = {}
|
||
i = 0
|
||
n = len(body)
|
||
while i < n:
|
||
while i < n and body[i] in " \t\n,":
|
||
i += 1
|
||
if i >= n:
|
||
break
|
||
m = re.compile(r"(\w+)\s*=").match(body, i)
|
||
if not m:
|
||
break
|
||
fname = m.group(1).lower()
|
||
i = m.end()
|
||
while i < n and body[i] in " \t\n":
|
||
i += 1
|
||
if i >= n:
|
||
break
|
||
if body[i] == "{":
|
||
depth = 1
|
||
j = i + 1
|
||
while j < n and depth:
|
||
if body[j] == "{":
|
||
depth += 1
|
||
elif body[j] == "}":
|
||
depth -= 1
|
||
j += 1
|
||
value = body[i + 1 : j - 1]
|
||
i = j
|
||
else:
|
||
m2 = re.compile(r"([^,]+)").match(body, i)
|
||
if not m2:
|
||
i += 1
|
||
continue
|
||
value = m2.group(1)
|
||
i = m2.end()
|
||
value = value.strip()
|
||
value = re.sub(r"\s*\n\s*", " ", value)
|
||
value = value.replace("{", "").replace("}", "").strip()
|
||
value = _clean_bib_value(value)
|
||
fields[fname] = value
|
||
return fields
|
||
|
||
|
||
def _clean_bib_value(value: str) -> str:
|
||
"""Remove escapes LaTeX correntes em .bib (diacríticos, \\& etc.).
|
||
|
||
Ex.: ``Ki\\vs\\vs`` -> ``Kiss``, ``L'opez`` -> ``Lopez``,
|
||
``Tirke\\cs`` -> ``Tirkes``.
|
||
"""
|
||
value = re.sub(
|
||
r"\\[\'\"~vc](?:\{([A-Za-z])\}|([A-Za-z]))",
|
||
lambda m: m.group(1) or m.group(2),
|
||
value,
|
||
flags=re.I,
|
||
)
|
||
for esc, ch in (("\\&", "&"), ("\\$", "$"), ("\\%", "%"),
|
||
("\\_", "_"), ("\\#", "#")):
|
||
value = value.replace(esc, ch)
|
||
value = value.replace("---", " — ").replace("--", " – ")
|
||
return value
|
||
|
||
|
||
def parse_bib(bib_path: Path | str) -> dict[str, Reference]:
|
||
text = Path(bib_path).read_text(encoding="utf-8")
|
||
refs: dict[str, Reference] = {}
|
||
for etype, key, body in _split_entries(text):
|
||
if not key:
|
||
continue
|
||
refs[key] = Reference(key=key, type=etype, fields=_parse_fields(body))
|
||
return refs
|