Projeto analise-artigo: análise ABNT de artigos LaTeX
- texparse.py: extração de estrutura abntex2 (título, autores, resumo, palavras-chave, seções, citações, figuras), sem dependências externas - bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos - analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch adaptado pt-BR e heurísticas de prosa - report.py: relatório Markdown + JSON - artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
This commit is contained in:
167
analise_artigo/bib.py
Normal file
167
analise_artigo/bib.py
Normal file
@@ -0,0 +1,167 @@
|
||||
"""Parser simples de arquivos .bib (BibTeX).
|
||||
|
||||
Usa apenas a biblioteca padrão; suporta campos com valores em chaves
|
||||
aninhadas e valores multi-linha.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
@dataclass
|
||||
class Reference:
|
||||
key: str
|
||||
type: str # article, inproceedings, misc, ...
|
||||
fields: dict[str, str] = field(default_factory=dict)
|
||||
|
||||
@property
|
||||
def authors(self) -> str:
|
||||
value = self.fields.get("author", "").strip()
|
||||
# "X, A. and B, C. and others" -> "X, A., B, C. et al." (apenas campos author)
|
||||
value = re.sub(r"\s+and\s+others\b", " et al.", value, flags=re.I)
|
||||
return re.sub(r"\s+and\s+", ", ", value, flags=re.I)
|
||||
|
||||
@property
|
||||
def title(self) -> str:
|
||||
return self.fields.get("title", "").strip()
|
||||
|
||||
@property
|
||||
def year(self) -> str:
|
||||
return self.fields.get("year", "").strip()
|
||||
|
||||
|
||||
@property
|
||||
def journal(self) -> str:
|
||||
return self.fields.get(
|
||||
"journal", self.fields.get("booktitle", "")
|
||||
).strip()
|
||||
|
||||
def label(self) -> str:
|
||||
"""Rótulo curto autor (ano)."""
|
||||
first = re.split(r",|and", self.authors)[0].strip()
|
||||
if "," in first:
|
||||
surname = first.split(",")[0].strip()
|
||||
else:
|
||||
surname = last_token(first)
|
||||
return f"{surname} ({self.year})" if self.year else surname
|
||||
|
||||
|
||||
def last_token(name: str) -> str:
|
||||
parts = name.replace("-", " ").split()
|
||||
return parts[-1] if parts else name
|
||||
|
||||
|
||||
def _split_entries(text: str) -> list[tuple[str, str, str]]:
|
||||
"""Retorna [(type, key, body), ...] respeitando aninhamento de chaves."""
|
||||
entries: list[tuple[str, str, str]] = []
|
||||
i = 0
|
||||
while True:
|
||||
m = re.compile(r"@(\w+)\s*\{").search(text, i)
|
||||
if not m:
|
||||
break
|
||||
etype, key = m.group(1), ""
|
||||
start = m.end()
|
||||
depth = 1
|
||||
j = start
|
||||
while j < len(text) and depth:
|
||||
if text[j] == "{":
|
||||
depth += 1
|
||||
elif text[j] == "}":
|
||||
depth -= 1
|
||||
j += 1
|
||||
body = text[start : j - 1]
|
||||
|
||||
# chave = conteúdo até a primeira vírgula fora de chaves
|
||||
depth = 0
|
||||
k = 0
|
||||
while k < len(body):
|
||||
c = body[k]
|
||||
if c == "{":
|
||||
depth += 1
|
||||
elif c == "}":
|
||||
depth -= 1
|
||||
elif c == "," and depth == 0:
|
||||
break
|
||||
if c == "\n" and depth == 0:
|
||||
break
|
||||
k += 1
|
||||
key = body[:k].strip()
|
||||
entries.append((etype, key, body[k + 1 :]))
|
||||
i = j
|
||||
return entries
|
||||
|
||||
|
||||
def _parse_fields(body: str) -> dict[str, str]:
|
||||
fields: dict[str, str] = {}
|
||||
i = 0
|
||||
n = len(body)
|
||||
while i < n:
|
||||
while i < n and body[i] in " \t\n,":
|
||||
i += 1
|
||||
if i >= n:
|
||||
break
|
||||
m = re.compile(r"(\w+)\s*=").match(body, i)
|
||||
if not m:
|
||||
break
|
||||
fname = m.group(1).lower()
|
||||
i = m.end()
|
||||
while i < n and body[i] in " \t\n":
|
||||
i += 1
|
||||
if i >= n:
|
||||
break
|
||||
if body[i] == "{":
|
||||
depth = 1
|
||||
j = i + 1
|
||||
while j < n and depth:
|
||||
if body[j] == "{":
|
||||
depth += 1
|
||||
elif body[j] == "}":
|
||||
depth -= 1
|
||||
j += 1
|
||||
value = body[i + 1 : j - 1]
|
||||
i = j
|
||||
else:
|
||||
m2 = re.compile(r"([^,]+)").match(body, i)
|
||||
if not m2:
|
||||
i += 1
|
||||
continue
|
||||
value = m2.group(1)
|
||||
i = m2.end()
|
||||
value = value.strip()
|
||||
value = re.sub(r"\s*\n\s*", " ", value)
|
||||
value = value.replace("{", "").replace("}", "").strip()
|
||||
value = _clean_bib_value(value)
|
||||
fields[fname] = value
|
||||
return fields
|
||||
|
||||
|
||||
def _clean_bib_value(value: str) -> str:
|
||||
"""Remove escapes LaTeX correntes em .bib (diacríticos, \\& etc.).
|
||||
|
||||
Ex.: ``Ki\\vs\\vs`` -> ``Kiss``, ``L'opez`` -> ``Lopez``,
|
||||
``Tirke\\cs`` -> ``Tirkes``.
|
||||
"""
|
||||
value = re.sub(
|
||||
r"\\[\'\"~vc](?:\{([A-Za-z])\}|([A-Za-z]))",
|
||||
lambda m: m.group(1) or m.group(2),
|
||||
value,
|
||||
flags=re.I,
|
||||
)
|
||||
for esc, ch in (("\\&", "&"), ("\\$", "$"), ("\\%", "%"),
|
||||
("\\_", "_"), ("\\#", "#")):
|
||||
value = value.replace(esc, ch)
|
||||
value = value.replace("---", " — ").replace("--", " – ")
|
||||
return value
|
||||
|
||||
|
||||
def parse_bib(bib_path: Path | str) -> dict[str, Reference]:
|
||||
text = Path(bib_path).read_text(encoding="utf-8")
|
||||
refs: dict[str, Reference] = {}
|
||||
for etype, key, body in _split_entries(text):
|
||||
if not key:
|
||||
continue
|
||||
refs[key] = Reference(key=key, type=etype, fields=_parse_fields(body))
|
||||
return refs
|
||||
Reference in New Issue
Block a user