Commit dd2c57ce authored by Andoni Jimenez's avatar Andoni Jimenez
Browse files

Merge branch 'rename-gui-texts' into 'main'

Rename voids -> groups, update pd_concat,  add missing columns, changs input...

See merge request !6
parents 2c832eaa fea9cbf2
Loading
Loading
Loading
Loading
+3 −4
Original line number Diff line number Diff line
@@ -13,18 +13,17 @@ VERSION = '0.1.2'

if __name__ == '__main__':
    args = utils.getOptions(VERSION)
    selectedFiles, selectedVoids = utils.getFiles()
    selectedFiles, selectedGroups = utils.getFiles()

    if len(selectedFiles) > 0:
        if args.list:
            print(selectedFiles.filename)

        else:
            print('INFO:\t' + str(len(selectedFiles)) + ' galaxies found '+
                  'in selected voids.')
            print(f"INFO:\t {str(len(selectedFiles))} galaxies found in selected groups.")
            df = utils.expand_df(selectedFiles)
            app = QtWidgets.QApplication(sys.argv)
            window = gui.Ui(df, selectedVoids)
            window = gui.Ui(df, selectedGroups)
            app.exec_()
    else:
        print('ERROR: No files found with the arguments given.')

galaxies.csv

0 → 100755
+31 −0
Original line number Diff line number Diff line
group,galaxy,ra,dec,filename
1,15,210.927048,-1.137346,image_15.png
1,254,211.020782,0.998166,image_254.png
1,368,209.399231,-0.143221,image_368.png
1,1432,131.115585,51.447163,image_1432.png
2,2053,229.601944,32.742622,image_2053.png
2,4578,176.255005,14.47608,image_4578.png
1,6809,212.720978,1.369397,image_6809.png
1,11563,174.978165,3.054553,image_11563.png
2,12345,205.853745,12.518266,image_12345.png
4,15248,201.117096,18.23148,image_15248.png
2,15420,194.252426,49.074078,image_15420.png
2,18520,232.43869,27.118929,image_18520.png
4,20763,192.414612,12.625293,image_20763.png
1,23654,140.026764,47.221436,image_23654.png
2,23658,181.49881,10.360068,image_23658.png
4,28650,179.265366,21.407429,image_28650.png
1,32874,177.717056,4.518072,image_32874.png
4,42136,190.907364,17.933109,image_42136.png
1,45125,134.666763,49.30088,image_45125.png
4,46302,148.82164,24.927872,image_46302.png
4,52413,192.979187,26.558268,image_52413.png
2,54852,233.504868,29.229906,image_54852.png
4,63246,169.911362,44.340736,image_63246.png
2,65321,219.951035,41.046577,image_65321.png
2,76469,168.293427,39.167614,image_76469.png
2,79852,238.548569,21.56811,image_79852.png
1,87502,140.757462,51.277439,image_87502.png
2,91645,182.419159,48.642597,image_91645.png
1,98652,142.558029,46.853474,image_98652.png
4,100000,227.259003,24.918394,image_100000.png
+13 −13
Original line number Diff line number Diff line
@@ -8,16 +8,16 @@ import sys


class Ui(QtWidgets.QMainWindow):
    def __init__(self, df, selectedVoids):
    def __init__(self, df, selectedGroups):
        super(Ui, self).__init__()
        uic.loadUi(str(Path('src')/Path('gui.ui')), self)

        self.title = 'CAVIssify ' + utils.getVersion()
        self.title = 'GALAssify ' + utils.getVersion()
        self.setWindowIcon(QIcon(str(Path('res/window_icon.png'))))

        # Load data frame and make it accessible along the class:
        self.df = df
        self.voids = selectedVoids
        self.galaxies = selectedGroups

        # Find buttons:
        self.pb_prev = self.findChild(QtWidgets.QPushButton, 'pb_prev')
@@ -93,10 +93,10 @@ class Ui(QtWidgets.QMainWindow):
    def fillList(self, showAllSavedData=False):
        #self.fileList.setRowCount(len(self.df.index))
        self.fileList.setColumnCount(6)
        self.fileList.setHorizontalHeaderLabels(['Void',
        self.fileList.setHorizontalHeaderLabels(['Group',
                                                 'Galaxy',
                                                 'Processed',
                                                 'RA',
                                                 'Ra',
                                                 'Dec',
                                                 'Filename'])
        self.fileList.horizontalHeaderItem(2).setTextAlignment(Qt.AlignHCenter)
@@ -119,7 +119,7 @@ class Ui(QtWidgets.QMainWindow):
        for i, row in self.df.iterrows():
            # Place void data in integer format:
            self.fileList.setItem(i, 0, QtWidgets.QTableWidgetItem())
            self.fileList.item(i, 0).setData(Qt.DisplayRole, int(row['void']))
            self.fileList.item(i, 0).setData(Qt.DisplayRole, int(row['group']))

            # Place galaxy data in integer format:
            self.fileList.setItem(i, 1, QtWidgets.QTableWidgetItem())
@@ -137,7 +137,7 @@ class Ui(QtWidgets.QMainWindow):
            # Place filename in the row:
            self.fileList.setItem(i, 5, QtWidgets.QTableWidgetItem(str(row['filename'])))

            if (not showAllSavedData) and (int(row['void']) not in self.voids):
            if (not showAllSavedData) and (int(row['group']) not in self.galaxies):
                self.fileList.hideRow(i)

        # Order by RA coordinate:
@@ -200,10 +200,10 @@ class Ui(QtWidgets.QMainWindow):
            item = self.df.loc[self.df['filename'] == fn]
            self.imgPath = item['fullpath'].item()

            v = str(item['void'].item())
            g = str(item['galaxy'].item())
            self.setWindowTitle('V: ' + v +
                                ' | G: ' + g +
            grp = str(item['group'].item())
            gal = str(item['galaxy'].item())
            self.setWindowTitle('Grp: ' + grp +
                                ' | Gal: ' + gal +
                                ' | ' + self.title)

            self.fileList.setCellWidget(index, 2,
@@ -259,8 +259,8 @@ class Ui(QtWidgets.QMainWindow):
    def toggleAIDVisibility(self):
        showAllSavedData = self.act_allSavedData.isChecked()
        for i in range(self.fileList.rowCount()):
            rowVoid = self.fileList.item(i, 0).text()
            if (not showAllSavedData) and (int(rowVoid) not in self.voids):
            row = self.fileList.item(i, 0).text()
            if (not showAllSavedData) and (int(row) not in self.galaxies):
                self.fileList.hideRow(i)
            else:
                self.fileList.showRow(i)
+1 −1
Original line number Diff line number Diff line
@@ -583,7 +583,7 @@
  </widget>
  <action name="action_vn">
   <property name="text">
    <string>Void number...</string>
    <string>Group number...</string>
   </property>
  </action>
  <action name="action_folder">
+72 −55
Original line number Diff line number Diff line
@@ -13,13 +13,13 @@ from typing import Union
args = None
VERSION = ''
importData = pd.DataFrame()
voids = pd.DataFrame()
groups = pd.DataFrame()

def getOptions(version):
    global args
    global VERSION

    parser = argparse.ArgumentParser(prog='CAVYssify', description="CAVYssify: Tool to manually classify void galaxies.")
    parser = argparse.ArgumentParser(prog='GALAssify', description="GALAssify: Tool to manually classify galaxies.")

    parser.add_argument('--version', action='version', version='%(prog)s ' + version,
                        help="Prints software version and exit.\n")
@@ -29,17 +29,17 @@ def getOptions(version):
    #                    required=not('-p' in sys.argv
    #                    or '--path' in sys.argv
    #                    or '--version' in sys.argv), type=dir_file,
    #                    help="Image or list to classify. Not required if path or path + void are given.\n")
    #                    help="Image or list to classify. Not required if path or path + group are given.\n")
    parser.add_argument('-s', '--savefile',
                        #required=not('--version' in sys.argv),
                        default='output.csv',
                        help="CSV file to load and export changes. If does not exists, a new one is created.\n")
    parser.add_argument("-l", "--list", action="store_true",
                        help="List selected files only and exit.\n")
    parser.add_argument('-vf', '--voidsfile', default='voids.csv',
                        help="Voids database file in *.csv format.\n")
    parser.add_argument('void', metavar='VOID', type=int, nargs='+',
                        help="Void number. Selects images with name format: *_<void>_*_*.png\n")
    parser.add_argument('-vf', '--inputfile', default='galaxies.csv',
                        help="Galaxy database file in *.csv format.\n")
    parser.add_argument('group', metavar='GROUP', type=int, nargs='*',
                        help="Group number. Selects images with name format: *_<group>_*_*.png\n")

    args = parser.parse_args()
    VERSION = version
@@ -53,28 +53,33 @@ def getVersion():

def getFiles():
    global args
    global voids
    files = []
    selectedVoids = []
    selectedFiles = pd.DataFrame()
    global groups
    # files = []
    selectedGroups = []
    selectedFiles = pd.DataFrame(columns = groups.columns)

    if args.path:
        if Path(args.voidsfile).is_file():
            voids = readVoidsFile()
        if Path(args.inputfile).is_file():
            groups = readInputFile()
        else:
            voids = createVoidsFile()
            groups = createInputFile()

        availableGroups = groups.group.unique()
        print(f"INFO:\tAvailable groups: {str(availableGroups)}")

        availableVoids = voids.void.unique()
        print('INFO:\tAvailable voids: ' + str(availableVoids))
        
        if len(args.void) > 0:
            for void in args.void:
                if void in availableVoids:
                    selectedFiles = pd_concat(selectedFiles, voids[voids.void == void])
                    #selectedFiles = selectedFiles.append(voids[voids.void == void ])
                    selectedVoids.append(int(void))
        if len(args.group) > 0:
            for group in args.group:
                if group in availableGroups:
                    selectedFiles = pd_concat(selectedFiles, groups[groups.group == group])
                    selectedGroups.append(int(group))
                else:
                    print(f'WARNING:\tGroup {group} not available.')
        else:
                    print('WARNING:\tVoid ', void, 'not available.')
            print(f'INFO:\tNo subgroup selected. Using all available by default.')
            selectedFiles = groups.copy()
            selectedGroups = availableGroups
                
        #else:
        #    formats = ['*.png']

@@ -88,8 +93,7 @@ def getFiles():

    #else:
    #    files = []

    return selectedFiles, selectedVoids
    return selectedFiles, selectedGroups


def dir_path(path):
@@ -109,49 +113,51 @@ def dir_file(file):
        raise argparse.ArgumentTypeError("readable_file:" + file
                                         + " is not a valid file.")

VFCOLUMNS = ['void','galaxy','ra','dec','filename']
VFCOLUMNS = ['group','galaxy','ra','dec','filename']

def readVoidsFile():
    print('INFO:\tReading from voids.csv file... ', end='', flush=True)
    voids = pd.read_csv(args.voidsfile,
def readInputFile():
    fname = args.inputfile
    print(f'INFO:\tReading from {fname} file... ', end='', flush=True)
    groups = pd.read_csv(fname,
                        converters={
                                     'void': int,
                                     'group': int,
                                     'galaxy': int,
                                     'ra': float,
                                     'dec': float,
                                    }
                        )
    voids = voids.sort_values(by=['void', 'ra', 'galaxy'])
    groups = groups.sort_values(by=['group', 'ra', 'galaxy'])
    print('Done!')
    return voids
    return groups

def createVoidsFile():
    print('INFO:\tCreating voids.csv file... ', end='', flush=True)
    voids = pd.DataFrame(columns=VFCOLUMNS)
def createInputFile():
    fname = args.inputfile
    print(f'INFO:\tCreating {fname} file... ', end='', flush=True)
    groups = pd.DataFrame(columns=VFCOLUMNS)
    if args.path:
        for file in Path(args.path).glob('pan_*_*_ppak.png'):
            entry = {
                        'void': int(file.stem.split('_')[1]),
                        'group': int(file.stem.split('_')[1]),
                        'galaxy': int(file.stem.split('_')[2]),
                        'ra':float(0),
                        'dec':float(0),
                        'filename': str(file.name),
                    }
            voids = pd_concat(voids, entry)
            #voids = voids.append(entry, ignore_index=True)
    voids = voids.sort_values(by=['void','galaxy'])
    voids.to_csv(args.voidsfile, columns=VFCOLUMNS, index=False)
            groups = pd_concat(groups, entry)
            #groups = groups.append(entry, ignore_index=True)
    groups = groups.sort_values(by=['group','galaxy'])
    groups.to_csv(fname, columns=VFCOLUMNS, index=False)
    print('Done!')
    return voids
    return groups


### PANDAS UTILS


COLUMNS = ["filename", "void", "galaxy", "morphology", "large", "tiny", "faceon",
COLUMNS = ["filename", "group", "galaxy", "morphology", "large", "tiny", "faceon",
           "edgeon", "star", "calibration", "recentre", "duplicated",
           "member", "hiiregion", "yes", "no", "comment", "processed",
           "fullpath"]
           "fullpath", "ra", "dec"]

MORPHOLOGY = ['elliptical', 'spiral', 'irregular', 'other', '']

@@ -169,11 +175,11 @@ def getRadioButtonsMorphology():


def getCheckBoxesColumns():
    return COLUMNS[4:-3]
    return COLUMNS[4:-5]


def getExportableColumns():
    return COLUMNS[:-2]
    return COLUMNS[:-4]


def checkColumnsMismatch(importDataColumns):
@@ -194,11 +200,11 @@ def checkColumnsMismatch(importDataColumns):
def expand_df(selectedFiles):
    global args
    global importData
    global voids
    global groups
    if Path(args.savefile).is_file():
        importData = pd.read_csv(args.savefile,
                                 converters={
                                             'void': int,
                                             'group': int,
                                             'galaxy': int,
                                            }
                                )
@@ -227,8 +233,8 @@ def expand_df(selectedFiles):
            processedUnselectedData = importData[importData.fullpath == '']
            for i, row in processedUnselectedData.iterrows():
                file = Path(args.path) / Path(row['filename'])
                ra = voids.loc[voids.galaxy == row.galaxy].ra.item()
                dec = voids.loc[voids.galaxy == row.galaxy].dec.item()
                ra = groups.loc[groups.galaxy == row.galaxy].ra.item()
                dec = groups.loc[groups.galaxy == row.galaxy].dec.item()
                importData.loc[importData.galaxy == row.galaxy, 'fullpath'] = file.absolute()
                importData.loc[importData.galaxy == row.galaxy, 'ra'] = ra
                importData.loc[importData.galaxy == row.galaxy, 'dec'] = dec
@@ -248,7 +254,7 @@ def newEntry(row):
    file = Path(args.path) / Path(row['filename'])
    entry = {
        'filename': file.name,
        'void': row.void,
        'group': row.group,
        'galaxy': row.galaxy,
        'morphology': MORPHOLOGY[-1],
    }
@@ -279,14 +285,14 @@ def save_df(df):
    #                            df.loc[df['processed'] == True]])

    # Remove old values, keep last ones:
    exportData = processedItems.drop_duplicates(['void','galaxy'],keep='last').sort_values('void')
    exportData = processedItems.drop_duplicates(['group','galaxy'],keep='last').sort_values('group')

    # Export final dataframe:
    exportData.to_csv(args.savefile, columns=getExportableColumns(),
                          index=False)
    
def pd_concat(df: pd.DataFrame, data: Union[pd.DataFrame, list, dict]) -> pd.DataFrame:
        """ Concats data in an array to the given dataframe
        """ Concats data to the given dataframe

        Parameters
        ----------
@@ -296,7 +302,9 @@ def pd_concat(df: pd.DataFrame, data: Union[pd.DataFrame, list, dict]) -> pd.Dat
        data: list, dict or pandas.Dataframe
            Data to be concatenated to the input pandas Dataframe.
            - List of the values to be concatenated (order of input values and Dataframe columns must match).
            - Dict of the key:values, where keys match the Dataframe columns.
            - Dict of the 'key:values', where keys match the Dataframe columns (if not, values are put to NaN).
            - pandas.Dataframe where columns (should) match the input Dataframe 
              (if not, new columns are created or values are put to NaN).

        Returns
        -------
@@ -309,9 +317,18 @@ def pd_concat(df: pd.DataFrame, data: Union[pd.DataFrame, list, dict]) -> pd.Dat
            df_data = pd.DataFrame([data], columns=df.columns)
        elif type(data) == dict:
            if len(data) != len(df.columns):
                warnings.warn('Input data [dict] missing input dataframe keys. Missing values insterted as NaN')
                warnings.warn('Input data [dict] missing input dataframe keys. Missing values inserted as NaN')
                print(list(data.keys()))
                print(list(df.columns))
            df_data = pd.DataFrame([data])
        elif type(data) == pd.DataFrame:
            if not df.empty:
                cmp_df_data = list(df.keys()[~df.keys().isin(data.keys())])
                cmp_data_df = list(data.keys()[~data.keys().isin(df.keys())])
                if len(cmp_df_data) > 0:
                    warnings.warn(f'Input data missing input dataframe column. {cmp_df_data}')
                if len(cmp_data_df) > 0:
                    warnings.warn(f'Input data column(s) not in input dataframe. {cmp_data_df}')
            df_data = data
            
        df = pd.concat([df, df_data], ignore_index=True)