Forgot to import an exception I used

[htsworkflow.git] / htsworkflow / submission / submission.py
diff --git a/htsworkflow/submission/submission.py b/htsworkflow/submission/submission.py

index 98c25d57befc4047c4cc846ecd72f26fe02c740d..77a1b68e4db763ab83eb50135f9613e8264ce973 100644 (file)
--- a/htsworkflow/submission/submission.py
+++ b/htsworkflow/submission/submission.py
@@ -19,21 +19,22 @@ from htsworkflow.util.rdfhelp import \
       toTypedNode, \
       fromTypedNode
  from htsworkflow.util.hashfile import make_md5sum
-
+from htsworkflow.submission.fastqname import FastqName
  from htsworkflow.submission.daf import \
       MetadataLookupException, \
+     ModelException, \
       get_submission_uri
  
-logger = logging.getLogger(__name__)
+LOGGER = logging.getLogger(__name__)
  
  class Submission(object):
-    def __init__(self, name, model):
+    def __init__(self, name, model, host):
          self.name = name
          self.model = model
  
          self.submissionSet = get_submission_uri(self.name)
-        self.submissionSetNS = RDF.NS(str(self.submissionSet) + '/')
-        self.libraryNS = RDF.NS('http://jumpgate.caltech.edu/library/')
+        self.submissionSetNS = RDF.NS(str(self.submissionSet) + '#')
+        self.libraryNS = RDF.NS('{0}/library/'.format(host))
  
          self.__view_map = None
  
@@ -41,11 +42,11 @@ class Submission(object):
          """Examine files in our result directory
          """
          for lib_id, result_dir in result_map.items():
-            logger.info("Importing %s from %s" % (lib_id, result_dir))
+            LOGGER.info("Importing %s from %s" % (lib_id, result_dir))
              try:
                  self.import_analysis_dir(result_dir, lib_id)
              except MetadataLookupException, e:
-                logger.error("Skipping %s: %s" % (lib_id, str(e)))
+                LOGGER.error("Skipping %s: %s" % (lib_id, str(e)))
  
      def import_analysis_dir(self, analysis_dir, library_id):
          """Import a submission directories and update our model as needed
@@ -57,7 +58,8 @@ class Submission(object):
  
          submission_files = os.listdir(analysis_dir)
          for filename in submission_files:
-            self.construct_file_attributes(analysis_dir, libNode, filename)
+            pathname = os.path.abspath(os.path.join(analysis_dir, filename))
+            self.construct_file_attributes(analysis_dir, libNode, pathname)
  
      def construct_file_attributes(self, analysis_dir, libNode, pathname):
          """Looking for the best extension
@@ -68,18 +70,28 @@ class Submission(object):
          """
          path, filename = os.path.split(pathname)
  
-        logger.debug("Searching for view")
-        file_classification = self.find_best_match(filename)
-        if file_classification is None:
-            logger.warn("Unrecognized file: {0}".format(pathname))
+        LOGGER.debug("Searching for view")
+        file_type = self.find_best_match(filename)
+        if file_type is None:
+            LOGGER.warn("Unrecognized file: {0}".format(pathname))
              return None
-        if str(file_classification) == str(libraryOntology['ignore']):
+        if str(file_type) == str(libraryOntology['ignore']):
              return None
  
          an_analysis_name = self.make_submission_name(analysis_dir)
          an_analysis = self.get_submission_node(analysis_dir)
          an_analysis_uri = str(an_analysis.uri)
+        file_classification = self.model.get_target(file_type,
+                                                    rdfNS['type'])
+        if file_classification is None:
+            errmsg = 'Could not find class for {0}'
+            LOGGER.warning(errmsg.format(str(file_type)))
+            return
  
+        self.model.add_statement(
+            RDF.Statement(self.submissionSetNS[''],
+                          submissionOntology['has_submission'],
+                          an_analysis))
          self.model.add_statement(RDF.Statement(an_analysis,
                                                 submissionOntology['name'],
                                                 toTypedNode(an_analysis_name)))
@@ -91,7 +103,7 @@ class Submission(object):
                                                 submissionOntology['library'],
                                                 libNode))
  
-        logger.debug("Adding statements to {0}".format(str(an_analysis)))
+        LOGGER.debug("Adding statements to {0}".format(str(an_analysis)))
          # add track specific information
          self.model.add_statement(
              RDF.Statement(an_analysis,
@@ -103,17 +115,21 @@ class Submission(object):
                            an_analysis))
  
          # add file specific information
-        fileNode = self.link_file_to_classes(filename,
-                                             an_analysis,
-                                             an_analysis_uri,
-                                             analysis_dir)
+        fileNode = self.make_file_node(pathname, an_analysis)
          self.add_md5s(filename, fileNode, analysis_dir)
+        self.add_fastq_metadata(filename, fileNode)
+        self.model.add_statement(
+            RDF.Statement(fileNode,
+                          rdfNS['type'],
+                          file_type))
+        LOGGER.debug("Done.")
  
-        logger.debug("Done.")
-
-    def link_file_to_classes(self, filename, submissionNode, submission_uri, analysis_dir):
+    def make_file_node(self, pathname, submissionNode):
+        """Create file node and attach it to its submission.
+        """
          # add file specific information
-        fileNode = RDF.Node(RDF.Uri(submission_uri + '/' + filename))
+        path, filename = os.path.split(pathname)
+        fileNode = RDF.Node(RDF.Uri('file://'+ os.path.abspath(pathname)))
          self.model.add_statement(
              RDF.Statement(submissionNode,
                            dafTermOntology['has_file'],
@@ -125,24 +141,74 @@ class Submission(object):
          return fileNode
  
      def add_md5s(self, filename, fileNode, analysis_dir):
-        logger.debug("Updating file md5sum")
+        LOGGER.debug("Updating file md5sum")
          submission_pathname = os.path.join(analysis_dir, filename)
          md5 = make_md5sum(submission_pathname)
          if md5 is None:
              errmsg = "Unable to produce md5sum for {0}"
-            logger.warning(errmsg.format(submission_pathname))
+            LOGGER.warning(errmsg.format(submission_pathname))
          else:
              self.model.add_statement(
                  RDF.Statement(fileNode, dafTermOntology['md5sum'], md5))
  
+    def add_fastq_metadata(self, filename, fileNode):
+        # How should I detect if this is actually a fastq file?
+        try:
+            fqname = FastqName(filename=filename)
+        except ValueError:
+            # currently its just ignore it if the fastq name parser fails
+            return
+        
+        terms = [('flowcell', libraryOntology['flowcell_id']),
+                 ('lib_id', libraryOntology['library_id']),
+                 ('lane', libraryOntology['lane_number']),
+                 ('read', libraryOntology['read']),
+                 ('cycle', libraryOntology['read_length'])]
+        for file_term, model_term in terms:
+            value = fqname.get(file_term)
+            if value is not None:
+                s = RDF.Statement(fileNode, model_term, toTypedNode(value))
+                self.model.append(s)
+
      def _add_library_details_to_model(self, libNode):
+        # attributes that can have multiple values
+        set_attributes = set((libraryOntology['has_lane'],
+                              libraryOntology['has_mappings'],
+                              dafTermOntology['has_file']))
          parser = RDF.Parser(name='rdfa')
          new_statements = parser.parse_as_stream(libNode.uri)
+        toadd = []
          for s in new_statements:
+            # always add "collections"
+            if s.predicate in set_attributes:
+                toadd.append(s)
+                continue
              # don't override things we already have in the model
              targets = list(self.model.get_targets(s.subject, s.predicate))
              if len(targets) == 0:
-                self.model.append(s)
+                toadd.append(s)
+
+        for s in toadd:
+            self.model.append(s)
+
+        self._add_lane_details(libNode)
+
+    def _add_lane_details(self, libNode):
+        """Import lane details
+        """
+        query = RDF.Statement(libNode, libraryOntology['has_lane'], None)
+        lanes = []
+        for lane_stmt in self.model.find_statements(query):
+            lanes.append(lane_stmt.object)
+
+        parser = RDF.Parser(name='rdfa')
+        for lane in lanes:
+            LOGGER.debug("Importing %s" % (lane.uri,))
+            try:
+                parser.parse_into_model(self.model, lane.uri)
+            except RDF.RedlandError, e:
+                LOGGER.error("Error accessing %s" % (lane.uri,))
+                raise e
  
  
      def find_best_match(self, filename):
@@ -178,11 +244,11 @@ class Submission(object):
          for s in self.model.find_statements(filename_query):
              view_name = s.subject
              literal_re = s.object.literal_value['string']
-            logger.debug("Found: %s" % (literal_re,))
+            LOGGER.debug("Found: %s" % (literal_re,))
              try:
                  filename_re = re.compile(literal_re)
              except re.error, e:
-                logger.error("Unable to compile: %s" % (literal_re,))
+                LOGGER.error("Unable to compile: %s" % (literal_re,))
              patterns[literal_re] = view_name
          return patterns
  
@@ -254,3 +320,17 @@ class Submission(object):
                  "Unrecognized library type %s for %s" % \
                  (library_type, str(libNode)))
  
+    def execute_query(self, template, context):
+        """Execute the query, returning the results
+        """
+        formatted_query = template.render(context)
+        LOGGER.debug(formatted_query)
+        query = RDF.SPARQLQuery(str(formatted_query))
+        rdfstream = query.execute(self.model)
+        results = []
+        for record in rdfstream:
+            d = {}
+            for key, value in record.items():
+                d[key] = fromTypedNode(value)
+            results.append(d)
+        return results