+ public static void preparePhylogenyForParsimonyAnalyses( final Phylogeny intree,
+ final String[][] input_file_properties ) {
+ final String[] genomes = new String[ input_file_properties.length ];
+ for( int i = 0; i < input_file_properties.length; ++i ) {
+ if ( intree.getNodes( input_file_properties[ i ][ 1 ] ).size() > 1 ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, "node named [" + input_file_properties[ i ][ 1 ]
+ + "] is not unique in input tree " + intree.getName() );
+ }
+ genomes[ i ] = input_file_properties[ i ][ 1 ];
+ }
+ //
+ final PhylogenyNodeIterator it = intree.iteratorPostorder();
+ while ( it.hasNext() ) {
+ final PhylogenyNode n = it.next();
+ if ( ForesterUtil.isEmpty( n.getName() ) ) {
+ if ( n.getNodeData().isHasTaxonomy()
+ && !ForesterUtil.isEmpty( n.getNodeData().getTaxonomy().getTaxonomyCode() ) ) {
+ n.setName( n.getNodeData().getTaxonomy().getTaxonomyCode() );
+ }
+ else if ( n.getNodeData().isHasTaxonomy()
+ && !ForesterUtil.isEmpty( n.getNodeData().getTaxonomy().getScientificName() ) ) {
+ n.setName( n.getNodeData().getTaxonomy().getScientificName() );
+ }
+ else if ( n.getNodeData().isHasTaxonomy()
+ && !ForesterUtil.isEmpty( n.getNodeData().getTaxonomy().getCommonName() ) ) {
+ n.setName( n.getNodeData().getTaxonomy().getCommonName() );
+ }
+ else {
+ ForesterUtil
+ .fatalError( surfacing.PRG_NAME,
+ "node with no name, scientific name, common name, or taxonomy code present" );
+ }
+ }
+ }
+ //
+ final List<String> igns = PhylogenyMethods.deleteExternalNodesPositiveSelection( genomes, intree );
+ if ( igns.size() > 0 ) {
+ System.out.println( "Not using the following " + igns.size() + " nodes:" );
+ for( int i = 0; i < igns.size(); ++i ) {
+ System.out.println( " " + i + ": " + igns.get( i ) );
+ }
+ System.out.println( "--" );
+ }
+ for( final String[] input_file_propertie : input_file_properties ) {
+ try {
+ intree.getNode( input_file_propertie[ 1 ] );
+ }
+ catch ( final IllegalArgumentException e ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, "node named [" + input_file_propertie[ 1 ]
+ + "] not present/not unique in input tree" );
+ }
+ }
+ }
+
+ public static void printOutPercentageOfMultidomainProteins( final SortedMap<Integer, Integer> all_genomes_domains_per_potein_histo,
+ final Writer log_writer ) {
+ int sum = 0;
+ for( final Entry<Integer, Integer> entry : all_genomes_domains_per_potein_histo.entrySet() ) {
+ sum += entry.getValue();
+ }
+ final double percentage = ( 100.0 * ( sum - all_genomes_domains_per_potein_histo.get( 1 ) ) ) / sum;
+ ForesterUtil.programMessage( surfacing.PRG_NAME, "Percentage of multidomain proteins: " + percentage + "%" );
+ log( "Percentage of multidomain proteins: : " + percentage + "%", log_writer );
+ }
+
+ public static void processFilter( final File filter_file, final SortedSet<String> filter ) {
+ SortedSet<String> filter_str = null;
+ try {
+ filter_str = ForesterUtil.file2set( filter_file );
+ }
+ catch ( final IOException e ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, e.getMessage() );
+ }
+ if ( filter_str != null ) {
+ for( final String string : filter_str ) {
+ filter.add( string );
+ }
+ }
+ if ( surfacing.VERBOSE ) {
+ System.out.println( "Filter:" );
+ for( final String domainId : filter ) {
+ System.out.println( domainId );
+ }
+ }
+ }
+
+ public static String[][] processInputGenomesFile( final File input_genomes ) {
+ String[][] input_file_properties = null;
+ try {
+ input_file_properties = ForesterUtil.file22dArray( input_genomes );
+ }
+ catch ( final IOException e ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME,
+ "genomes files is to be in the following format \"<hmmpfam output file> <species>\": "
+ + e.getLocalizedMessage() );
+ }
+ final Set<String> specs = new HashSet<String>();
+ final Set<String> paths = new HashSet<String>();
+ for( int i = 0; i < input_file_properties.length; ++i ) {
+ if ( !PhyloXmlUtil.TAXOMONY_CODE_PATTERN.matcher( input_file_properties[ i ][ 1 ] ).matches() ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, "illegal format for species code: "
+ + input_file_properties[ i ][ 1 ] );
+ }
+ if ( specs.contains( input_file_properties[ i ][ 1 ] ) ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, "species code " + input_file_properties[ i ][ 1 ]
+ + " is not unique" );
+ }
+ specs.add( input_file_properties[ i ][ 1 ] );
+ if ( paths.contains( input_file_properties[ i ][ 0 ] ) ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, "path " + input_file_properties[ i ][ 0 ]
+ + " is not unique" );
+ }
+ paths.add( input_file_properties[ i ][ 0 ] );
+ final String error = ForesterUtil.isReadableFile( new File( input_file_properties[ i ][ 0 ] ) );
+ if ( !ForesterUtil.isEmpty( error ) ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, error );
+ }
+ }
+ return input_file_properties;
+ }
+
+ public static void processPlusMinusAnalysisOption( final CommandLineArguments cla,
+ final List<String> high_copy_base,
+ final List<String> high_copy_target,
+ final List<String> low_copy,
+ final List<Object> numbers ) {
+ if ( cla.isOptionSet( surfacing.PLUS_MINUS_ANALYSIS_OPTION ) ) {
+ if ( !cla.isOptionValueSet( surfacing.PLUS_MINUS_ANALYSIS_OPTION ) ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, "no value for 'plus-minus' file: -"
+ + surfacing.PLUS_MINUS_ANALYSIS_OPTION + "=<file>" );
+ }
+ final File plus_minus_file = new File( cla.getOptionValue( surfacing.PLUS_MINUS_ANALYSIS_OPTION ) );
+ final String msg = ForesterUtil.isReadableFile( plus_minus_file );
+ if ( !ForesterUtil.isEmpty( msg ) ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, "can not read from \"" + plus_minus_file + "\": " + msg );
+ }
+ processPlusMinusFile( plus_minus_file, high_copy_base, high_copy_target, low_copy, numbers );
+ }
+ }
+
+ // First numbers is minimal difference, second is factor.
+ public static void processPlusMinusFile( final File plus_minus_file,
+ final List<String> high_copy_base,
+ final List<String> high_copy_target,
+ final List<String> low_copy,
+ final List<Object> numbers ) {
+ Set<String> species_set = null;
+ int min_diff = surfacing.PLUS_MINUS_ANALYSIS_MIN_DIFF_DEFAULT;
+ double factor = surfacing.PLUS_MINUS_ANALYSIS_FACTOR_DEFAULT;
+ try {
+ species_set = ForesterUtil.file2set( plus_minus_file );
+ }
+ catch ( final IOException e ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, e.getMessage() );
+ }
+ if ( species_set != null ) {
+ for( final String species : species_set ) {
+ final String species_trimmed = species.substring( 1 );
+ if ( species.startsWith( "+" ) ) {
+ if ( low_copy.contains( species_trimmed ) ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME,
+ "species/genome names can not appear with both '+' and '-' suffix, as appears the case for: \""
+ + species_trimmed + "\"" );
+ }
+ high_copy_base.add( species_trimmed );
+ }
+ else if ( species.startsWith( "*" ) ) {
+ if ( low_copy.contains( species_trimmed ) ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME,
+ "species/genome names can not appear with both '*' and '-' suffix, as appears the case for: \""
+ + species_trimmed + "\"" );
+ }
+ high_copy_target.add( species_trimmed );
+ }
+ else if ( species.startsWith( "-" ) ) {
+ if ( high_copy_base.contains( species_trimmed ) || high_copy_target.contains( species_trimmed ) ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME,
+ "species/genome names can not appear with both '+' or '*' and '-' suffix, as appears the case for: \""
+ + species_trimmed + "\"" );
+ }
+ low_copy.add( species_trimmed );
+ }
+ else if ( species.startsWith( "$D" ) ) {
+ try {
+ min_diff = Integer.parseInt( species.substring( 3 ) );
+ }
+ catch ( final NumberFormatException e ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME,
+ "could not parse integer value for minimal difference from: \""
+ + species.substring( 3 ) + "\"" );
+ }
+ }
+ else if ( species.startsWith( "$F" ) ) {
+ try {
+ factor = Double.parseDouble( species.substring( 3 ) );
+ }
+ catch ( final NumberFormatException e ) {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, "could not parse double value for factor from: \""
+ + species.substring( 3 ) + "\"" );
+ }
+ }
+ else if ( species.startsWith( "#" ) ) {
+ // Comment, ignore.
+ }
+ else {
+ ForesterUtil
+ .fatalError( surfacing.PRG_NAME,
+ "species/genome names in 'plus minus' file must begin with '*' (high copy target genome), '+' (high copy base genomes), '-' (low copy genomes), '$D=<integer>' minimal Difference (default is 1), '$F=<double>' factor (default is 1.0), double), or '#' (ignore) suffix, encountered: \""
+ + species + "\"" );
+ }
+ numbers.add( new Integer( min_diff + "" ) );
+ numbers.add( new Double( factor + "" ) );
+ }
+ }
+ else {
+ ForesterUtil.fatalError( surfacing.PRG_NAME, "'plus minus' file [" + plus_minus_file + "] appears empty" );
+ }
+ }
+
+ /*
+ * species | protein id | n-terminal domain | c-terminal domain | n-terminal domain per domain E-value | c-terminal domain per domain E-value
+ *
+ *