import re
from collections import defaultdict
_TOKEN_RE = re.compile(r"([A-Z][a-z]?)(\d+\.?\d*)?")
_FULL_FLAT_RE = re.compile(r"^([A-Z][a-z]?\d*\.?\d*)+$")
_VALID_ELEMENTS = {
"H", "He", "Li", "Be", "B", "C", "N", "O", "F", "Ne", "Na", "Mg", "Al", "Si", "P", "S", "Cl",
"Ar", "K", "Ca", "Sc", "Ti", "V", "Cr", "Mn", "Fe", "Co", "Ni", "Cu", "Zn", "Ga", "Ge", "As",
"Se", "Br", "Kr", "Rb", "Sr", "Y", "Zr", "Nb", "Mo", "Tc", "Ru", "Rh", "Pd", "Ag", "Cd", "In",
"Sn", "Sb", "Te", "I", "Xe", "Cs", "Ba", "La", "Ce", "Pr", "Nd", "Pm", "Sm", "Eu", "Gd", "Tb",
"Dy", "Ho", "Er", "Tm", "Yb", "Lu", "Hf", "Ta", "W", "Re", "Os", "Ir", "Pt", "Au", "Hg", "Tl",
"Pb", "Bi", "Po", "At", "Rn", "Fr", "Ra", "Ac", "Th", "Pa", "U", "Np", "Pu", "Am", "Cm", "Bk",
"Cf", "Es", "Fm", "Md", "No", "Lr", "Rf", "Db", "Sg", "Bh", "Hs", "Mt", "Ds", "Rg", "Cn", "Nh",
"Fl", "Mc", "Lv", "Ts", "Og",
}
def parse_flat_formula(formula):
if not formula or not _FULL_FLAT_RE.match(formula):
return None
amounts = {}
for symbol, amount_str in _TOKEN_RE.findall(formula):
if symbol not in _VALID_ELEMENTS:
return None
amount = float(amount_str) if amount_str else 1.0
if amount <= 0.0:
return None
if symbol in amounts:
return None amounts[symbol] = amount
if not amounts:
return None
return amounts
def element_set(parsed):
return set(parsed.keys())
def _smoke_check():
cases = {
"Ti3SiC2": {"Ti": 3.0, "Si": 1.0, "C": 2.0},
"Fe0.9Ni0.1": {"Fe": 0.9, "Ni": 0.1},
"CaTiO3": {"Ca": 1.0, "Ti": 1.0, "O": 3.0},
"Bi2Te3": {"Bi": 2.0, "Te": 3.0},
"Ag": {"Ag": 1.0},
}
for formula, expected in cases.items():
got = parse_flat_formula(formula)
assert got == expected, f"{formula}: expected {expected}, got {got}"
for bad in ["(PbS)1.18(TiS2)2", "DyCl3ยท6H2O", "NaCl-KCl", "TiTi3", "", "Xx2O3"]:
assert parse_flat_formula(bad) is None, f"expected None for {bad!r}, got {parse_flat_formula(bad)!r}"
assert parse_flat_formula("CoNi2O") == {"Co": 1.0, "Ni": 2.0, "O": 1.0} print("parse_flat_formula smoke check: OK")
if __name__ == "__main__":
_smoke_check()