Python: supprimer le double dans une colonne spécifique
df = df.drop_duplicates(subset=['Column1', 'Column2'], keep='first')
Andrea Perlato
df = df.drop_duplicates(subset=['Column1', 'Column2'], keep='first')
import pandas as pd
# making data frame from csv file
data = pd.read_csv("employees.csv")
# sorting by first name
data.sort_values("First Name", inplace = True)
# dropping ALL duplicte values
data.drop_duplicates(subset ="First Name",keep = False, inplace = True)
# displaying data
print(data)
df.drop_duplicates(['A','B'],keep= 'last')
df = df.drop_duplicates(subset=['Column1', 'Column2'], keep='first')
# Exemple
import pandas as pd
df = pd.DataFrame({"A":["foo", "foo", "foo", "bar"], "B":[0,1,1,1], "C":["A","A","B","A"]})
df.drop_duplicates(subset=['A', 'C'], keep=False)
df = df.drop_duplicates(subset=['Column1', 'Column2'], keep='first')
# Exemple
import pandas as pd
df = pd.DataFrame({"A":["foo", "foo", "foo", "bar"], "B":[0,1,1,1], "C":["A","A","B","A"]})
df.drop_duplicates(subset=['A', 'C'], keep=False)
df = df.loc[:,~df.columns.duplicated()]