Projeto analise-artigo: análise ABNT de artigos LaTeX

- texparse.py: extração de estrutura abntex2 (título, autores, resumo,
  palavras-chave, seções, citações, figuras), sem dependências externas
- bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos
- analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch
  adaptado pt-BR e heurísticas de prosa
- report.py: relatório Markdown + JSON
- artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
This commit is contained in:
2026-09-25 13:24:56 -03:00
commit 550440f4a7
14 changed files with 1591 additions and 0 deletions

167
analise_artigo/bib.py Normal file
View File

@@ -0,0 +1,167 @@
"""Parser simples de arquivos .bib (BibTeX).
Usa apenas a biblioteca padrão; suporta campos com valores em chaves
aninhadas e valores multi-linha.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from pathlib import Path
@dataclass
class Reference:
key: str
type: str # article, inproceedings, misc, ...
fields: dict[str, str] = field(default_factory=dict)
@property
def authors(self) -> str:
value = self.fields.get("author", "").strip()
# "X, A. and B, C. and others" -> "X, A., B, C. et al." (apenas campos author)
value = re.sub(r"\s+and\s+others\b", " et al.", value, flags=re.I)
return re.sub(r"\s+and\s+", ", ", value, flags=re.I)
@property
def title(self) -> str:
return self.fields.get("title", "").strip()
@property
def year(self) -> str:
return self.fields.get("year", "").strip()
@property
def journal(self) -> str:
return self.fields.get(
"journal", self.fields.get("booktitle", "")
).strip()
def label(self) -> str:
"""Rótulo curto autor (ano)."""
first = re.split(r",|and", self.authors)[0].strip()
if "," in first:
surname = first.split(",")[0].strip()
else:
surname = last_token(first)
return f"{surname} ({self.year})" if self.year else surname
def last_token(name: str) -> str:
parts = name.replace("-", " ").split()
return parts[-1] if parts else name
def _split_entries(text: str) -> list[tuple[str, str, str]]:
"""Retorna [(type, key, body), ...] respeitando aninhamento de chaves."""
entries: list[tuple[str, str, str]] = []
i = 0
while True:
m = re.compile(r"@(\w+)\s*\{").search(text, i)
if not m:
break
etype, key = m.group(1), ""
start = m.end()
depth = 1
j = start
while j < len(text) and depth:
if text[j] == "{":
depth += 1
elif text[j] == "}":
depth -= 1
j += 1
body = text[start : j - 1]
# chave = conteúdo até a primeira vírgula fora de chaves
depth = 0
k = 0
while k < len(body):
c = body[k]
if c == "{":
depth += 1
elif c == "}":
depth -= 1
elif c == "," and depth == 0:
break
if c == "\n" and depth == 0:
break
k += 1
key = body[:k].strip()
entries.append((etype, key, body[k + 1 :]))
i = j
return entries
def _parse_fields(body: str) -> dict[str, str]:
fields: dict[str, str] = {}
i = 0
n = len(body)
while i < n:
while i < n and body[i] in " \t\n,":
i += 1
if i >= n:
break
m = re.compile(r"(\w+)\s*=").match(body, i)
if not m:
break
fname = m.group(1).lower()
i = m.end()
while i < n and body[i] in " \t\n":
i += 1
if i >= n:
break
if body[i] == "{":
depth = 1
j = i + 1
while j < n and depth:
if body[j] == "{":
depth += 1
elif body[j] == "}":
depth -= 1
j += 1
value = body[i + 1 : j - 1]
i = j
else:
m2 = re.compile(r"([^,]+)").match(body, i)
if not m2:
i += 1
continue
value = m2.group(1)
i = m2.end()
value = value.strip()
value = re.sub(r"\s*\n\s*", " ", value)
value = value.replace("{", "").replace("}", "").strip()
value = _clean_bib_value(value)
fields[fname] = value
return fields
def _clean_bib_value(value: str) -> str:
"""Remove escapes LaTeX correntes em .bib (diacríticos, \\& etc.).
Ex.: ``Ki\\vs\\vs`` -> ``Kiss``, ``L'opez`` -> ``Lopez``,
``Tirke\\cs`` -> ``Tirkes``.
"""
value = re.sub(
r"\\[\'\"~vc](?:\{([A-Za-z])\}|([A-Za-z]))",
lambda m: m.group(1) or m.group(2),
value,
flags=re.I,
)
for esc, ch in (("\\&", "&"), ("\\$", "$"), ("\\%", "%"),
("\\_", "_"), ("\\#", "#")):
value = value.replace(esc, ch)
value = value.replace("---", " — ").replace("--", " – ")
return value
def parse_bib(bib_path: Path | str) -> dict[str, Reference]:
text = Path(bib_path).read_text(encoding="utf-8")
refs: dict[str, Reference] = {}
for etype, key, body in _split_entries(text):
if not key:
continue
refs[key] = Reference(key=key, type=etype, fields=_parse_fields(body))
return refs